{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Simple notebook demonstrating training and submission of a time-series model\n\n<span style=\"color:red\">Pipeline for variable length sequences!</span>","metadata":{}},{"cell_type":"markdown","source":"<a id=\"contents\"></a>\n# Contents\n1. [Load a predefined dataset](#section-one)\n2. [Define the model](#section-two)\n3. [Training](#section-three)\n4. [Conversion to TFLite](#section-four)\n5. [Submission](#section-five)\n6. [Sample predictions](#section-six)","metadata":{}},{"cell_type":"code","source":"# As always, we need to do our imports first\nimport tensorflow as tf\nfrom tensorflow.keras import layers, optimizers\nimport numpy as np\nimport pandas as pd\nimport json\n\n\n# and define some constants\n# We will drop most of the face landmarks to reduce the dimensionality of the data\nLANDMARK_IDX = [0,9,11,13,14,17,117,118,119,199,346,347,348] + list(range(468,543))\nDATA_PATH = \"/kaggle/input/saved-tfdataset-of-google-isl-recognition-data/GoogleISLDatasetBatched\"\nDS_CARDINALITY = 185\nVAL_SIZE  = 18\nN_SIGNS = 250\nROWS_PER_FRAME = 543","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:20.274923Z","iopub.execute_input":"2023-03-04T18:44:20.275297Z","iopub.status.idle":"2023-03-04T18:44:28.759823Z","shell.execute_reply.started":"2023-03-04T18:44:20.275238Z","shell.execute_reply":"2023-03-04T18:44:28.758708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata = pd.read_csv(\"/kaggle/input/asl-signs/train.csv\")\nwith open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\") as f:\n    sign_map = json.load(f)\nsign_list = list(sign_map.keys())","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:28.762346Z","iopub.execute_input":"2023-03-04T18:44:28.763126Z","iopub.status.idle":"2023-03-04T18:44:28.958100Z","shell.execute_reply.started":"2023-03-04T18:44:28.763087Z","shell.execute_reply":"2023-03-04T18:44:28.957050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# 1. Loading the data\n\nWe use a saved tf.Dataset containing ragged batches of the competition data, generated in the same way that the submission data is loaded.\n\nRefer to [my notebook](https://www.kaggle.com/code/aapokossi/how-to-save-parquet-data-as-ragged-tf-dataset) for an explanation of how the dataset was generated and feel free to either use [my presaved Dataset](https://www.kaggle.com/datasets/aapokossi/saved-tfdataset-of-google-isl-recognition-data) or generate your own with any additional compute-intense preprocessing you desire!","metadata":{}},{"cell_type":"code","source":"# Let's do some simple preprocessing to make training faster and save memory\ndef preprocess(ragged_batch, labels):\n    \n    # drop most of the face mesh\n    ragged_batch = tf.gather(ragged_batch, LANDMARK_IDX, axis=2)\n\n    #fill nan\n    ragged_batch = tf.where(tf.math.is_nan(ragged_batch), tf.zeros_like(ragged_batch), ragged_batch)\n\n    #flatten landmark xyz coordinates and return\n    return tf.concat([ragged_batch[...,i] for i in range(3)],-1), labels","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:28.959601Z","iopub.execute_input":"2023-03-04T18:44:28.959947Z","iopub.status.idle":"2023-03-04T18:44:28.967959Z","shell.execute_reply.started":"2023-03-04T18:44:28.959910Z","shell.execute_reply":"2023-03-04T18:44:28.966931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data and split to train, validation\ndataset = tf.data.Dataset.load(DATA_PATH)\ndataset = dataset.map(preprocess)\nval_ds = dataset.take(VAL_SIZE).cache().prefetch(tf.data.AUTOTUNE)\ntrain_ds = dataset.skip(VAL_SIZE).cache().shuffle(20).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:28.971320Z","iopub.execute_input":"2023-03-04T18:44:28.971805Z","iopub.status.idle":"2023-03-04T18:44:32.505337Z","shell.execute_reply.started":"2023-03-04T18:44:28.971777Z","shell.execute_reply":"2023-03-04T18:44:32.503789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n# 2. Defining the classifier model","metadata":{}},{"cell_type":"code","source":"def get_callbacks():\n    return [\n            tf.keras.callbacks.EarlyStopping(\n            monitor=\"val_accuracy\",\n            patience=8,\n            restore_best_weights=True\n        ),\n        tf.keras.callbacks.ReduceLROnPlateau(\n            monitor = \"val_accuracy\",\n            factor = 0.5,\n            patience = 3\n        ),\n    ]","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:32.506762Z","iopub.execute_input":"2023-03-04T18:44:32.507157Z","iopub.status.idle":"2023-03-04T18:44:32.515731Z","shell.execute_reply.started":"2023-03-04T18:44:32.507117Z","shell.execute_reply":"2023-03-04T18:44:32.514729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dense_block(units, name):\n    fc = layers.Dense(units)\n    norm = layers.LayerNormalization()\n    act = layers.Activation(\"relu\")\n    return lambda x: act(norm(fc(x)))","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:32.517136Z","iopub.execute_input":"2023-03-04T18:44:32.517512Z","iopub.status.idle":"2023-03-04T18:44:32.525072Z","shell.execute_reply.started":"2023-03-04T18:44:32.517474Z","shell.execute_reply":"2023-03-04T18:44:32.524134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def classifier(lstm_units):\n    lstm = layers.LSTM(lstm_units)\n    out = layers.Dense(N_SIGNS, activation=\"softmax\")\n    return lambda x: out(lstm(x))","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:32.526535Z","iopub.execute_input":"2023-03-04T18:44:32.526969Z","iopub.status.idle":"2023-03-04T18:44:32.535133Z","shell.execute_reply.started":"2023-03-04T18:44:32.526933Z","shell.execute_reply":"2023-03-04T18:44:32.534221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder_units = [256,128,64]\nlstm_units = 128\n\n#define the inputs (ragged batches of time series of landmark coordinates)\ninputs = tf.keras.Input(shape=(None,3*len(LANDMARK_IDX)), ragged=True)\n\n# dense encoder model\nx = inputs\nfor i, n in enumerate(encoder_units):\n    x = dense_block(n, f\"encoder_{i}\")(x)\n\n# classifier model\nout = classifier(lstm_units)(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs=out)\n\nsteps_per_epoch = DS_CARDINALITY - VAL_SIZE\nboundaries = [steps_per_epoch * n for n in [20,30,40]]\nvalues = [1e-3,1e-4,1e-5,1e-6]\nlr_sched = optimizers.schedules.PiecewiseConstantDecay(boundaries, values)\noptimizer = optimizers.Adam(lr_sched)\n\nmodel.compile(optimizer=optimizer,\n              loss=\"sparse_categorical_crossentropy\",\n              metrics=[\"accuracy\",\"sparse_top_k_categorical_accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:32.536549Z","iopub.execute_input":"2023-03-04T18:44:32.537014Z","iopub.status.idle":"2023-03-04T18:44:33.313478Z","shell.execute_reply.started":"2023-03-04T18:44:32.536979Z","shell.execute_reply":"2023-03-04T18:44:33.312437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three\"></a>\n# 3. Training","metadata":{}},{"cell_type":"code","source":"model.fit(train_ds,\n          validation_data = val_ds,\n          callbacks = get_callbacks(),\n          epochs = 50,\n         )","metadata":{"execution":{"iopub.status.busy":"2023-03-04T18:44:33.314960Z","iopub.execute_input":"2023-03-04T18:44:33.315337Z","iopub.status.idle":"2023-03-04T19:09:31.437019Z","shell.execute_reply.started":"2023-03-04T18:44:33.315296Z","shell.execute_reply":"2023-03-04T19:09:31.435989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:09:31.440707Z","iopub.execute_input":"2023-03-04T19:09:31.441005Z","iopub.status.idle":"2023-03-04T19:09:31.482061Z","shell.execute_reply.started":"2023-03-04T19:09:31.440978Z","shell.execute_reply":"2023-03-04T19:09:31.481259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four\"></a>\n# 4. TFLite inference model and conversion","metadata":{}},{"cell_type":"code","source":"def get_inference_model(model):\n    inputs = tf.keras.Input(shape=(543,3), name=\"inputs\")\n    \n    # drop most of the face mesh\n    x = tf.gather(inputs, LANDMARK_IDX, axis=1)\n\n    # fill nan\n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n\n    # flatten landmark xyz coordinates ()\n    x = tf.concat([x[...,i] for i in range(3)], -1)\n\n    x = tf.expand_dims(x,0)\n    \n    # call trained model\n    out = model(x)\n    \n    # explicitly name the final (identity) layer for the submission format\n    out = layers.Activation(\"linear\", name=\"outputs\")(out)\n    \n    inference_model = tf.keras.Model(inputs=inputs, outputs=out)\n    inference_model.compile(loss=\"sparse_categorical_crossentropy\",\n                            metrics=\"accuracy\")\n    return inference_model","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:09:31.483232Z","iopub.execute_input":"2023-03-04T19:09:31.483630Z","iopub.status.idle":"2023-03-04T19:09:44.696851Z","shell.execute_reply.started":"2023-03-04T19:09:31.483588Z","shell.execute_reply":"2023-03-04T19:09:44.695738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model = get_inference_model(model)\ninference_model.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:09:44.699063Z","iopub.execute_input":"2023-03-04T19:09:44.700309Z","iopub.status.idle":"2023-03-04T19:09:45.136325Z","shell.execute_reply.started":"2023-03-04T19:09:44.700265Z","shell.execute_reply":"2023-03-04T19:09:45.135514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five\"></a>\n# 5. Generating submission","metadata":{}},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(inference_model)\ntflite_model = converter.convert()\nmodel_path = \"model.tflite\"\n\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)\n!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:09:45.137397Z","iopub.execute_input":"2023-03-04T19:09:45.137941Z","iopub.status.idle":"2023-03-04T19:09:58.790821Z","shell.execute_reply.started":"2023-03-04T19:09:45.137911Z","shell.execute_reply":"2023-03-04T19:09:58.789582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-six\"></a>\n# 6. Demonstrate TFLite inference","metadata":{}},{"cell_type":"code","source":"data_dir = \"/kaggle/input/asl-signs/\"\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(data_dir + pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:09:58.793462Z","iopub.execute_input":"2023-03-04T19:09:58.793873Z","iopub.status.idle":"2023-03-04T19:09:58.802259Z","shell.execute_reply.started":"2023-03-04T19:09:58.793830Z","shell.execute_reply":"2023-03-04T19:09:58.799571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tflite-runtime\nimport tflite_runtime.interpreter as tflite\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:09:58.803689Z","iopub.execute_input":"2023-03-04T19:09:58.804935Z","iopub.status.idle":"2023-03-04T19:10:11.261793Z","shell.execute_reply.started":"2023-03-04T19:09:58.804895Z","shell.execute_reply":"2023-03-04T19:10:11.260643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interpreter = tflite.Interpreter(model_path)\nfound_signatures = list(interpreter.get_signature_list().keys())\n# if REQUIRED_SIGNATURE not in found_signatures:\n#     raise KernelEvalException('Required input signature not found.')\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\ny_trues = []\ny_preds = []\nfor i in range(1000):\n    data = load_relevant_data_subset(metadata.path[i])\n    output = prediction_fn(inputs=data)\n    sign_pred = np.argmax(output[\"outputs\"])\n    \n    y_trues.append(metadata.sign[i])\n    y_preds.append(sign_list[sign_pred])","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:10:11.264712Z","iopub.execute_input":"2023-03-04T19:10:11.265084Z","iopub.status.idle":"2023-03-04T19:10:28.834037Z","shell.execute_reply.started":"2023-03-04T19:10:11.265045Z","shell.execute_reply":"2023-03-04T19:10:28.832977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_num = [sign_map[sign] for sign in y_trues]\ny_prednum = [sign_map[sign] for sign in y_preds]\nconfmat = tf.math.confusion_matrix(y_num,y_prednum)\nplt.imshow(confmat, cmap=\"binary\")","metadata":{"execution":{"iopub.status.busy":"2023-03-04T19:10:28.835868Z","iopub.execute_input":"2023-03-04T19:10:28.836248Z","iopub.status.idle":"2023-03-04T19:10:29.138836Z","shell.execute_reply.started":"2023-03-04T19:10:28.836191Z","shell.execute_reply":"2023-03-04T19:10:29.137823Z"},"trusted":true},"execution_count":null,"outputs":[]}]}