{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis notebook is created for [Kaggle's Sign Langueage Classifier Competition](https://www.kaggle.com/competitions/asl-signs/code)\n\nIn this notebooks, two datasets are used:\n* **asl-signs**: This dataset is provided and contains landmark data that has already been pulled from raw video using MediaPipe. Within the train_landmark_files directory, there are a number of folders with IDs that identify which participant the data came from. Within each of these folders, there are parquet files that are uniquely identified by their sequence number.\n* **saved-tfdataset-...**: This dataset I borrowed from a Kaggle submission and is already cleaned a bit. I used it for training. Here is a link to the dataset: https://www.kaggle.com/datasets/aapokossi/saved-tfdataset-of-google-isl-recognition-data","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Import Libraries and Set File Directories","metadata":{}},{"cell_type":"code","source":"# import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow.keras import layers, optimizers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-02T01:37:44.024972Z","iopub.execute_input":"2023-05-02T01:37:44.025841Z","iopub.status.idle":"2023-05-02T01:37:51.033234Z","shell.execute_reply.started":"2023-05-02T01:37:44.025802Z","shell.execute_reply":"2023-05-02T01:37:51.032161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set files directories\nLANDMARK_FILES_DIR = \"/kaggle/input/asl-signs/train_landmark_files\"\nTRAIN_FILE = \"/kaggle/input/asl-signs/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-05-02T01:37:51.035889Z","iopub.execute_input":"2023-05-02T01:37:51.037401Z","iopub.status.idle":"2023-05-02T01:37:51.042233Z","shell.execute_reply.started":"2023-05-02T01:37:51.037361Z","shell.execute_reply":"2023-05-02T01:37:51.041281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n# Visualize data","metadata":{}},{"cell_type":"markdown","source":"Using a parquet data file format since it is much faster than using a traditional CSV.","metadata":{}},{"cell_type":"code","source":"# Reading the first few entries of data from a random parquet file to get a feel of the data\nsample = pd.read_parquet(\"/kaggle/input/asl-signs/train_landmark_files/16069/100015657.parquet\")\nsample.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T01:37:51.043595Z","iopub.execute_input":"2023-05-02T01:37:51.044033Z","iopub.status.idle":"2023-05-02T01:37:51.224731Z","shell.execute_reply.started":"2023-05-02T01:37:51.043996Z","shell.execute_reply":"2023-05-02T01:37:51.223743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three\"></a>\n# Load Data","metadata":{}},{"cell_type":"code","source":"# Set constants and pick important landmarks\nLANDMARK_IDX = [0,9,11,13,14,17,117,118,119,199,346,347,348] + list(range(468,543))\nDATA_PATH = \"/kaggle/input/saved-tfdataset-of-google-isl-recognition-data/GoogleISLDatasetBatched\"\nDS_CARDINALITY = 185\nVAL_SIZE  = 18\nN_SIGNS = 250\nROWS_PER_FRAME = 543","metadata":{"execution":{"iopub.status.busy":"2023-05-02T01:37:52.887540Z","iopub.execute_input":"2023-05-02T01:37:52.888675Z","iopub.status.idle":"2023-05-02T01:37:52.894687Z","shell.execute_reply.started":"2023-05-02T01:37:52.888635Z","shell.execute_reply":"2023-05-02T01:37:52.893640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To keep it simple, we will use the preprocessed [tf.Dataset](https://www.tensorflow.org/api_docs/python/tf/data/Dataset) from [tfdataset-of-google-isl-recognition-data](https://www.kaggle.com/datasets/aapokossi/saved-tfdataset-of-google-isl-recognition-data).","metadata":{}},{"cell_type":"code","source":"def preprocess(ragged_batch, labels):\n    ragged_batch = tf.gather(ragged_batch, LANDMARK_IDX, axis=2)\n    # replace NaN values with 0\n    ragged_batch = tf.where(tf.math.is_nan(ragged_batch), tf.zeros_like(ragged_batch), ragged_batch)\n    return tf.concat([ragged_batch[...,i] for i in range(3)],-1), labels\n\ndataset = tf.data.Dataset.load(DATA_PATH)\ndataset = dataset.map(preprocess)\nval_ds = dataset.take(VAL_SIZE).cache().prefetch(tf.data.AUTOTUNE)\ntrain_ds = dataset.skip(VAL_SIZE).cache().shuffle(20).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T01:37:52.896122Z","iopub.execute_input":"2023-05-02T01:37:52.897033Z","iopub.status.idle":"2023-05-02T01:37:56.504541Z","shell.execute_reply.started":"2023-05-02T01:37:52.896997Z","shell.execute_reply":"2023-05-02T01:37:56.503500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four\"></a>\n# Train Model","metadata":{}},{"cell_type":"markdown","source":"Now let us get to the fun part, training the model!","metadata":{}},{"cell_type":"code","source":"# Reduce the learning rate by half when we reach a plateau for 5 epochs\ndef get_callbacks():\n    return [\n        tf.keras.callbacks.ReduceLROnPlateau(\n            monitor = \"val_accuracy\",\n            factor = 0.5,\n            patience = 5\n        ),\n    ]\n\n# a single dense block followed by a normalization block and relu activation\ndef dense_block(units, name):\n    fc = layers.Dense(units)\n    norm = layers.LayerNormalization()\n    act = layers.Activation(\"relu\")\n    drop = layers.Dropout(0.1)\n    return lambda x: drop(act(norm(fc(x))))\n\n# the lstm block with the final dense block for the classification\ndef classifier(lstm_units):\n    lstm = layers.LSTM(lstm_units)\n    out = layers.Dense(N_SIGNS, activation=\"softmax\")\n    return lambda x: out(lstm(x))","metadata":{"execution":{"iopub.status.busy":"2023-05-02T01:37:56.506130Z","iopub.execute_input":"2023-05-02T01:37:56.506489Z","iopub.status.idle":"2023-05-02T01:37:56.514294Z","shell.execute_reply.started":"2023-05-02T01:37:56.506452Z","shell.execute_reply":"2023-05-02T01:37:56.512841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# choose the number of nodes per layer\nencoder_units = [256, 128] # tune this\nlstm_units = 175 # tune this\n\n#define the inputs (ragged batches of time series of landmark coordinates)\ninputs = tf.keras.Input(shape=(None,3*len(LANDMARK_IDX)), ragged=True)\n\n# dense encoder model\nx = inputs\nfor i, n in enumerate(encoder_units):\n    x = dense_block(n, f\"encoder_{i}\")(x)\n\n# classifier model\nout = classifier(lstm_units)(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs=out)\n# print out a summary of the model structure\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# optimizer\n# add a decreasing learning rate scheduler to help convergence\nsteps_per_epoch = DS_CARDINALITY - VAL_SIZE\nboundaries = [steps_per_epoch * n for n in [30,50,70]]\nvalues = [1e-3,1e-4,1e-5,1e-6]\nlr_sched = optimizers.schedules.PiecewiseConstantDecay(boundaries, values)\noptimizer = optimizers.Adam(lr_sched)\n\nmodel.compile(optimizer=optimizer,\n              loss=\"sparse_categorical_crossentropy\",\n              metrics=[\"accuracy\",\"sparse_top_k_categorical_accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2023-05-02T01:37:57.014046Z","iopub.execute_input":"2023-05-02T01:37:57.014427Z","iopub.status.idle":"2023-05-02T01:37:57.047270Z","shell.execute_reply.started":"2023-05-02T01:37:57.014389Z","shell.execute_reply":"2023-05-02T01:37:57.046135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit the model with 100 epochs iteration\nmodel.fit(train_ds,\n          validation_data = val_ds,\n          callbacks = get_callbacks(),\n          epochs = 100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}