Using tensorflow to run CGOL
Posted: April 11th, 2023, 5:39 am
I've finally managed to write CGOL iterator using tensorflow, which is the adaption of LifeAPI iterator.
You can run it on whatever hardware that supports tf. Like GPU, multi GPU, TPU, cluster computing etc. they all have acceleration for tf and bitwise operations are part of its specification. Soon there would appear a lot of dedicated hardware for this purpose (mainly for neural nets). So it's a good idea to move to higher level libraries, which are hardware agnostic, and most hardware is actually adapted to them.
On my laptop with 1060 GTX, it runs twice faster than on my CPU, with ~500K CGOL iterations/sec of 64x64 grid.
EDIT: I've run Tesla T4 on google colab, and it was 1.5M iter/sec. I guess if you have something like 3090 RTX you will get to ~5M.
You can run it on whatever hardware that supports tf. Like GPU, multi GPU, TPU, cluster computing etc. they all have acceleration for tf and bitwise operations are part of its specification. Soon there would appear a lot of dedicated hardware for this purpose (mainly for neural nets). So it's a good idea to move to higher level libraries, which are hardware agnostic, and most hardware is actually adapted to them.
On my laptop with 1060 GTX, it runs twice faster than on my CPU, with ~500K CGOL iterations/sec of 64x64 grid.
EDIT: I've run Tesla T4 on google colab, and it was 1.5M iter/sec. I guess if you have something like 3090 RTX you will get to ~5M.
Code: Select all
import tensorflow as tf
import numpy as np
def Add(b1, b0, val):
t_b1 = tf.bitwise.bitwise_and(b0, val)
b1.assign(tf.bitwise.bitwise_or(b1, t_b1))
b0.assign(tf.bitwise.bitwise_xor(b0, val))
def Add_Init(b1, b0, val):
b1.assign(tf.bitwise.bitwise_and(b0, val))
b0.assign(tf.bitwise.bitwise_xor(b0, val))
def Add1(b2, b1, b0, val):
t_b2 = tf.bitwise.bitwise_and(b0, val)
b2.assign(tf.bitwise.bitwise_or(b2, tf.bitwise.bitwise_and(t_b2, b1)))
b1.assign(tf.bitwise.bitwise_xor(b1, t_b2))
b0.assign(tf.bitwise.bitwise_xor(b0, val))
def Add_Init1(b2, b1, b0, val):
t_b2 = tf.bitwise.bitwise_and(b0, val)
b2.assign(t_b2 & b1)
b1.assign(tf.bitwise.bitwise_xor(b1, t_b2))
b0.assign(tf.bitwise.bitwise_xor(b0, val))
class Life64Layer(tf.keras.layers.Layer):
def __init__(self, **kwargs):
super(Life64Layer, self).__init__(**kwargs)
self.sum0 = tf.Variable(tf.zeros((1, K, N), dtype=tf.uint64), trainable=False)
self.sum1 = tf.Variable(tf.zeros((1, K, N), dtype=tf.uint64), trainable=False)
self.sum2 = tf.Variable(tf.zeros((1, K, N), dtype=tf.uint64), trainable=False)
def Evolve(self, temp, bU0, bU1, bB0, bB1):
self.sum0.assign(tf.bitwise.left_shift(temp, 1))
Add_Init(self.sum1, self.sum0, tf.bitwise.right_shift(temp, 1))
Add(self.sum1, self.sum0, bU0)
Add_Init(self.sum2, self.sum1, bU1)
Add1(self.sum2, self.sum1, self.sum0, bB0)
Add(self.sum2, self.sum1, bB1)
return tf.bitwise.bitwise_and(tf.bitwise.bitwise_and(tf.bitwise.invert(self.sum2), self.sum1), tf.bitwise.bitwise_or(temp, self.sum0))
def call(self, inputs):
state = inputs
l = tf.bitwise.left_shift(state, 1)
r = tf.bitwise.right_shift(state, 1)
bit0 = tf.bitwise.bitwise_xor(tf.bitwise.bitwise_xor(l, r), state)
l_or_r = tf.bitwise.bitwise_or(l, r)
l_and_r = tf.bitwise.bitwise_and(l, r)
state_and_l_or_r = tf.bitwise.bitwise_and(state, l_or_r)
bit1 = tf.bitwise.bitwise_or(state_and_l_or_r, l_and_r)
bU0 = tf.roll(bit0, shift=-1, axis=2)
bU1 = tf.roll(bit1, shift=-1, axis=2)
bB0 = tf.roll(bit0, shift=1, axis=2)
bB1 = tf.roll(bit1, shift=1, axis=2)
return self.Evolve(state, bU0, bU1, bB0, bB1)
def print_bits(arr):
N = arr.shape[0]
for i in range(N):
s = "{0:b}".format(arr[i])
s = s.zfill(64).replace('0', '_').replace('1', 'O')
print(s)
if i == N-1:
break
print()
K = 10000
N = 64
x = np.random.randint(0, 2**64, size=(1, K, N), dtype=np.uint64)
print_bits(x[0, 0])
# convert numpy array to tensor with uint64 data type
x_tensor = tf.convert_to_tensor(x, dtype=tf.uint64)
model = tf.keras.Sequential([Life64Layer(input_shape=(K, N), dtype=tf.uint64)] +
[Life64Layer(dtype=tf.uint64) for _ in range(10)])
import time
for k in range(100):
start_time = time.time()
y_pred = model.predict(x_tensor, batch_size = 32, verbose = False)
end_time = time.time()
elapsed_time = end_time - start_time
print("Elapsed time: {:.2f} seconds".format(elapsed_time))
print_bits(y_pred[0, 1])
x_tensor = y_pred