{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"We show how to convert pytorch gru to keras gru\nOnce it is in keras, you can convert to tflite.\n\nthe approach we take is to re-code keras gru in basic ops (linear, etc) in pytorch. \nonce we verify the operations and results are the same, we can manually cpied the pytorch weights to keras.\n\nthis is work still in progress ...\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"'''\n#misc \n\n# https://pytorch.org/docs/stable/generated/torch.nn.GRU.html\n# https://blog.floydhub.com/gru-with-pytorch/\n# https://github.com/georgeyiasemis/Recurrent-Neural-Networks-from-scratch-using-PyTorch/blob/main/rnncells.py\n\n'''\n\nin_dim=4\nout_dim=5\nB=2\nL=3","metadata":{"execution":{"iopub.status.busy":"2023-03-18T07:05:29.660917Z","iopub.execute_input":"2023-03-18T07:05:29.661382Z","iopub.status.idle":"2023-03-18T07:05:29.699229Z","shell.execute_reply.started":"2023-03-18T07:05:29.661342Z","shell.execute_reply":"2023-03-18T07:05:29.698021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n\n### first: pytorch gru in basic ops\n## good explanatiopn for pytorch implementation: https://d2l.ai/chapter_recurrent-modern/gru.html\n\nclass PytorchGRU(nn.Module):\n\tdef __init__(self,):\n\t\tsuper().__init__()\n\t\tself.gru = nn.GRU(\n\t\t\tin_dim, out_dim, bidirectional=False, batch_first=True\n\t\t)\n\n\tdef forward(self, x):\n\t\ty1, h1 = self.gru(x)\n\n\t\t#======================================================================\n\t\tif 1:\n\t\t\tB,L,dim = x.shape\n\t\t\th = torch.zeros(B, out_dim).to(x)\n\t\t\ty2 = []\n\n\t\t\tfor i in range(L):\n\t\t\t\tx_t = F.linear(x[:,i], self.gru.weight_ih_l0, bias=self.gru.bias_ih_l0)\n\t\t\t\th_t = F.linear(h, self.gru.weight_hh_l0, bias=self.gru.bias_hh_l0)\n\t\t\t\tx_reset, x_upd, x_new = x_t.chunk(3, 1)\n\t\t\t\th_reset, h_upd, h_new = h_t.chunk(3, 1)\n\n\t\t\t\treset_gate  = torch.sigmoid(x_reset + h_reset)\n\t\t\t\tupdate_gate = torch.sigmoid(x_upd + h_upd)\n\t\t\t\tnew_gate = torch.tanh(x_new + (reset_gate * h_new))\n\t\t\t\ty = update_gate * h + (1 - update_gate) * new_gate\n\t\t\t\ty2.append(y)\n\t\t\t\th = y\n\t\t\t\tzz=0\n\t\t\ty2 = torch.stack(y2, 1)\n\t\t\th2 = h\n\n\t\t\tif 1: #debug\n\t\t\t\tisclose0 = torch.all(torch.isclose(y1, y2, rtol=1e-05, atol=1e-08, equal_nan=False))\n\t\t\t\tisclose1 = torch.all(torch.isclose(h1, h2, rtol=1e-05, atol=1e-08, equal_nan=False))\n\n\t\t\t\tprint('pytorch compare y1,y2')\n\t\t\t\tprint(isclose0)\n\t\t\t\tprint('y1', y1.reshape(-1)[:5])\n\t\t\t\tprint('y2', y2.reshape(-1)[:5])\n\t\t\t\tprint('pytorch compare h1,h2')\n\t\t\t\tprint(isclose1)\n\t\t\t\tprint('h1', h1.reshape(-1)[:5])\n\t\t\t\tprint('h2', h2.reshape(-1)[:5])\n\t\t\t\tprint('')\n\n\t\t#======================================================================\n\n\t\treturn x\n\n\nx = np.random.rand(B,L,in_dim).astype(np.float32)\npgru = PytorchGRU()\ny,h = pgru(torch.from_numpy(x))\n","metadata":{"execution":{"iopub.status.busy":"2023-03-18T07:05:29.701638Z","iopub.execute_input":"2023-03-18T07:05:29.702246Z","iopub.status.idle":"2023-03-18T07:05:32.766827Z","shell.execute_reply.started":"2023-03-18T07:05:29.702203Z","shell.execute_reply":"2023-03-18T07:05:32.765496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try to convert from Keras to tflite first.  \nI do not know why only static shape is supported.  \nWhy B=None gives error for input_signature @tf.function().  \n\nIf you are a tflite user and know the reason, please leave a comment.","metadata":{}},{"cell_type":"code","source":"#let's try to convert from keras to tflite first\n\nimport tensorflow as tf \ntry:\n    import tflite_runtime\nexcept:\n    !pip install tflite-runtime\n\nimport tflite_runtime.interpreter as tflite   \nimport tflite_runtime\nprint(tflite_runtime.__version__)\n\nclass KerasGRU(tf.keras.layers.Layer):\n    def __init__(self, ):\n        super().__init__()\n        self.gru = tf.keras.layers.GRU(\n            units=out_dim, \n            dropout=0.0, \n            return_sequences=True, \n            reset_after=False,  ## pytorch uses True , for keras, you can use False or True\n        )\n\n    def call(self, x):\n        y1 = self.gru(x)\n        return y1\n\nclass TFModel(tf.Module):\n\tdef __init__(self):\n\t\tsuper(TFModel, self).__init__()\n\t\tself.net = KerasGRU()\n\t\tself.net.trainable = False\n\n\t@tf.function(input_signature=[\n\t\ttf.TensorSpec(shape=[B, L, in_dim], dtype=tf.float32, name='inputs') \n        #only static shape supported ???? by B=None gives error ???? \n\t])\n\tdef __call__(self, inputs):\n\t\ty = self.net(inputs)\n\t\treturn {'outputs': y} \n        \ndef keras_to_tflite():\n     \n    tfmodel = TFModel()\n    xyz = np.random.rand(B,  L, in_dim).astype(np.float32)\n    ouput = tfmodel(xyz)\n    print(ouput)\n    print('TFModel() ok')\n\n    tflite_file = 'model.tflite' \n    # tf.saved_model.save(tfmodel, tf_file, signatures={'serving_default': tfmodel.__call__})\n    # converter = tf.lite.TFLiteConverter.from_saved_model(tf_file)\n    converter = tf.lite.TFLiteConverter.from_keras_model(TFModel())\n    \n    # converter.target_spec.supported_ops = [\n    #     tf.lite.OpsSet.TFLITE_BUILTINS,  # enable TensorFlow Lite ops.\n    #     tf.lite.OpsSet.SELECT_TF_OPS  # enable TensorFlow ops.\n    # ]\n    # converter.experimental_new_converter = True\n    # converter.allow_custom_ops = True\n    \n    converter.optimizations = [tf.lite.Optimize.DEFAULT]\n    tf_lite_model = converter.convert()\n    with open(tflite_file, 'wb') as f:\n        f.write(tf_lite_model)\n    print('tflite convert() passed !!')\n\nkeras_to_tflite()","metadata":{"execution":{"iopub.status.busy":"2023-03-18T07:05:32.769062Z","iopub.execute_input":"2023-03-18T07:05:32.769783Z","iopub.status.idle":"2023-03-18T07:05:59.738710Z","shell.execute_reply.started":"2023-03-18T07:05:32.769727Z","shell.execute_reply":"2023-03-18T07:05:59.736974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from https://www.tensorflow.org/api_docs/python/tf/keras/layers/GRU,  \n\n\"There are two variants of the GRU implementation. The default one is based on v3 and has reset gate applied to hidden state before matrix multiplication. The other one is based on original and has the order reversed.\n\nThe second variant is compatible with CuDNNGRU (GPU-only) and allows inference on CPU. Thus it has separate biases for kernel and recurrent_kernel. To use this variant, set reset_after=True and recurrent_activation='sigmoid'.\"  \n\n<a href=\"https://ibb.co/TrmcznX\"><img src=\"https://i.ibb.co/wYMyDbH/Selection-999-1476.png\" alt=\"Selection-999-1476\" border=\"0\"></a>\n\nHere is the call graph of tflite for\ntop : reset_after=True    \nbottom : reset_after=False    \n\nThey aren't much difference, Hence actually there may not be much difference if you use \"pytorch (basic ops) to onnx to tflite\" or \"keras (layer) to tflite\". I also note that there is a tflite loop. I wonder if this is going to be faster than temporal conv1d or fully dense net, or transformer net (which has parallel computation).","metadata":{}},{"cell_type":"markdown","source":"<a href=\"https://ibb.co/Dpt67fZ\"><img src=\"https://i.ibb.co/NSskL2M/Selection-999-1475.png\" alt=\"Selection-999-1475\" border=\"0\"></a>\n<a href=\"https://ibb.co/M6BF37P\"><img src=\"https://i.ibb.co/wsc5x4p/Selection-999-1474.png\" alt=\"Selection-999-1474\" border=\"0\"></a>","metadata":{}},{"cell_type":"code","source":"### keras gru  \n# detail code of guru cell\n# see https://github.com/keras-team/keras/blob/v2.11.0/keras/layers/rnn/gru.py\n# others:\n# https://github.com/keras-team/keras/issues/8860 \n# https://github.com/keras-team/keras/issues/15915\n# https://stackoverflow.com/questions/72809642/how-to-interpret-get-weights-for-keras-gru\n\n\nimport tensorflow as tf\nclass KerasGRU(tf.keras.layers.Layer):\n    def __init__(self, ):\n        super().__init__()\n        self.gru = tf.keras.layers.GRU(\n            units=out_dim, dropout=0.0, return_sequences=True, bias_initializer='ones',\n\t        reset_after=False, ## use false !!!!\n        )\n\n    def call(self, x):\n        y1 = self.gru(x)\n\n        #---\n        if 1: # reset_after=False only\n            B,L,dim = x.shape\n            h = tf.zeros((B, out_dim))\n            y2 = []\n\n            weight_ih_l0 = kg.trainable_weights[0]\n            weight_hh_l0 = kg.trainable_weights[1]\n            bias   = kg.trainable_weights[2]\n\n            for i in range(L):\n                x_t = tf.matmul(x[:,i], weight_ih_l0)\n                x_t = tf.add(x_t, bias)\n                x_upd, x_reset,  x_new = tf.split(x_t, 3, 1)  #split order is different from pytorch\n\n                h_t = tf.matmul(h, weight_hh_l0)\n                ## h_t   = tf.add(h_t, bias)  ###no bias ???\n                h_upd, h_reset, h_new = tf.split(h_t, 3, 1)\n\n\n                reset_gate  = tf.nn.sigmoid(x_reset + h_reset)\n                update_gate = tf.nn.sigmoid(x_upd + h_upd)\n                new_gate = tf.nn.tanh(x_new + (reset_gate * h_new))\n\n                y = update_gate * h + (1 - update_gate) * new_gate\n\n                y2.append(y)  #y1[:,0]\n                h = y\n\n            y2 = tf.stack(y2, 1)\n\n            # zz=0 \n            if 1: #debug\n                isclose0 = tf.math.reduce_all(tf.experimental.numpy.allclose(y1, y2, rtol=1e-05, atol=1e-08, equal_nan=False))\n\n                print('keras compare y1,y2')\n                print(isclose0)\n                print('y1', tf.reshape(y1,-1)[:5])\n                print('y2', tf.reshape(y2,-1)[:5])\n                print('')\n \n        return y1\n\nx  = np.random.rand(B,L,in_dim)\nkg = KerasGRU()\ny  = kg(x)\n#print(y)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-18T07:05:59.744411Z","iopub.execute_input":"2023-03-18T07:05:59.744841Z","iopub.status.idle":"2023-03-18T07:05:59.856295Z","shell.execute_reply.started":"2023-03-18T07:05:59.744799Z","shell.execute_reply":"2023-03-18T07:05:59.854711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to be updated ...\n\n# now that we understand the equations and differences of pytorch and keras gru, we want \n# to write a pytorch layer that copmte the same as keras gru\n\n\nclass MyPytorchGRU(nn.Module):\n    def __init__(self,):\n        super().__init__()\n        self.gru = nn.GRU(\n            in_dim, out_dim, bidirectional=False, batch_first=True\n        )\n        # it might be asier to define your own parameters and remove nn.GRU() ....\n        # i am lazy here to use that of nn.GRU()\n\n    def forward(self, x):\n        if 1:\n            B,L,dim = x.shape\n            h = torch.zeros(B, out_dim).to(x)\n            y2 = []\n\n            for i in range(L):\n                x_t = F.linear(x[:,i], self.gru.weight_ih_l0, bias=self.gru.bias_ih_l0)\n                h_t = F.linear(h, self.gru.weight_hh_l0, bias=None)\n                x_upd, x_reset, x_new = x_t.chunk(3, 1)\n                h_upd, h_reset, h_new = h_t.chunk(3, 1)\n\n                reset_gate  = torch.sigmoid(x_reset + h_reset)\n                update_gate = torch.sigmoid(x_upd + h_upd)\n                new_gate = torch.tanh(x_new + (reset_gate * h_new))\n                y = update_gate * h + (1 - update_gate) * new_gate\n                y2.append(y)\n                h = y\n            y2 = torch.stack(y2, 1)\n            h2 = h\n\n        return y2\n\nclass MyKerasGRU(tf.keras.layers.Layer):\n    def __init__(self, ):\n        super().__init__()\n        self.gru = tf.keras.layers.GRU(\n            units=out_dim, dropout=0.0, return_sequences=True, bias_initializer='ones',\n            reset_after=False, ## use false !!!!\n        )\n\n    def call(self, x):\n        y1 = self.gru(x)\n        return y1\n\n#make random input\nx = np.random.rand(B,L,in_dim).astype(np.float32)\n\n\n# make some random weights\n# or use state dict to retrieve pytorch weights\n#weight_ih_l0 = np.zeros((in_dim, 3*out_dim)) #np.random.uniform(-1,1,(in_dim, 3*out_dim))\n#weight_hh_l0 = np.zeros((out_dim, 3*out_dim)) #np.random.uniform(-1,1,(out_dim,3*out_dim))\n#bias_ih_l0   = np.zeros((3*out_dim)) #np.random.uniform(-1,1,(3*out_dim))\nweight_ih_l0 = np.random.uniform(-1,1,(in_dim, 3*out_dim))\nweight_hh_l0 = np.random.uniform(-1,1,(out_dim,3*out_dim))\nbias_ih_l0   = np.random.uniform(-1,1,(3*out_dim))\n\n#set weights for pytorch\npgru = MyPytorchGRU()\npgru.gru.weight_ih_l0.data[...] = torch.from_numpy(weight_ih_l0.T)\npgru.gru.weight_hh_l0.data[...] = torch.from_numpy(weight_hh_l0.T)\npgru.gru.bias_ih_l0.data[...] = torch.from_numpy(bias_ih_l0)\nprint('set weight ok for pytorch!')\n\n#https://stackoverflow.com/questions/47183159/how-to-set-weights-in-keras-with-a-numpy-array\nkgru = MyKerasGRU()\ny = kgru(x)\n\nkgru.set_weights([\n\tweight_ih_l0,\n\tweight_hh_l0,\n\tbias_ih_l0\n])\nprint('set weight ok for keras!')\n\n\n###########################################\n## run and compare results\npy = pgru(torch.from_numpy(x))\nky = kgru(x)\n\npy = py.data.numpy()\nky = ky.numpy()\n\nprint('check first 5 ...')\nprint(ky.reshape(-1)[:5])\nprint(py.reshape(-1)[:5])\nprint('check last 5 ...')\nprint(ky.reshape(-1)[-5:])\nprint(py.reshape(-1)[-5:])\n\nzz=0","metadata":{"execution":{"iopub.status.busy":"2023-03-18T07:38:05.853156Z","iopub.execute_input":"2023-03-18T07:38:05.853633Z","iopub.status.idle":"2023-03-18T07:38:05.940503Z","shell.execute_reply.started":"2023-03-18T07:38:05.853589Z","shell.execute_reply":"2023-03-18T07:38:05.938293Z"},"trusted":true},"execution_count":null,"outputs":[]}]}