{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Sign Language Recognition (SLR) using Deep Neural Network (DNN)\n\nFor Google - Isolated Sign Language Recognition Competition (https://www.kaggle.com/competitions/asl-signs), I used DNN to develop an isolated sign language recognition model in this notebook. \n\nBecause the training records for this dataset contain a variety of frame counts, I will calculate the mean frame for each set of training records to make it simpler to get started and train more quickly. The input shape for this model will be (n, 253, 3) and the output shape (n, 250). It's really difficult to do during inference. To meet the requirements of this competition, I will develop an inference model to do data preprocessing such as calculating the mean frame of the test file using input shapes of None, 253, 3, and imputation. Ultimately, it will have an output shape of 250.\n\n","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Importing Necessary Packages","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport gc\nimport tensorflow as tf\nimport json\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:13.509753Z","iopub.execute_input":"2023-04-25T11:11:13.510145Z","iopub.status.idle":"2023-04-25T11:11:13.516620Z","shell.execute_reply.started":"2023-04-25T11:11:13.510110Z","shell.execute_reply":"2023-04-25T11:11:13.515316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting Configuration","metadata":{}},{"cell_type":"code","source":"class cg:\n    aggregation_data_path = \"../input/isolated-sign-language-aggregation-dataset/\"\n    data_path = \"../input/asl-signs/\"\n    quick_experiment = False\n    is_training = True\n    use_aggregation_dataset = True\n    num_classes = 250\n    rows_per_frame = 543","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:13.521766Z","iopub.execute_input":"2023-04-25T11:11:13.522647Z","iopub.status.idle":"2023-04-25T11:11:13.528890Z","shell.execute_reply.started":"2023-04-25T11:11:13.522604Z","shell.execute_reply":"2023-04-25T11:11:13.527864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now Utilities","metadata":{}},{"cell_type":"code","source":"def read_dict(file_path):\n    path = os.path.expanduser(file_path)\n    with open(path, \"r\") as f:\n        ck = json.load(f)\n    return ck\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / cg.rows_per_frame)\n    data = data.values.reshape(n_frames, cg.rows_per_frame, len(data_columns))\n    return data.astype(np.float32)\n\ndef load_relevant_data_subset_with_imputation(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    data.replace(np.nan, 0, inplace=True)\n    n_frames = int(len(data) / cg.rows_per_frame)\n    data = data.values.reshape(n_frames, cg.rows_per_frame, len(data_columns))\n    return data.astype(np.float32)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:13.531052Z","iopub.execute_input":"2023-04-25T11:11:13.531540Z","iopub.status.idle":"2023-04-25T11:11:13.540802Z","shell.execute_reply.started":"2023-04-25T11:11:13.531490Z","shell.execute_reply":"2023-04-25T11:11:13.539656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the Dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(f\"{cg.aggregation_data_path}train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:13.570823Z","iopub.execute_input":"2023-04-25T11:11:13.571112Z","iopub.status.idle":"2023-04-25T11:11:13.697366Z","shell.execute_reply.started":"2023-04-25T11:11:13.571085Z","shell.execute_reply":"2023-04-25T11:11:13.695984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 21 participants. Each of them create about 3000 to 5000 training records.","metadata":{}},{"cell_type":"code","source":"train.participant_id.nunique()\ntrain.participant_id.value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:13.699033Z","iopub.execute_input":"2023-04-25T11:11:13.699330Z","iopub.status.idle":"2023-04-25T11:11:14.011601Z","shell.execute_reply.started":"2023-04-25T11:11:13.699301Z","shell.execute_reply":"2023-04-25T11:11:14.010547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 94477 training samples in total.","metadata":{}},{"cell_type":"code","source":"label_index = read_dict(f\"{cg.data_path}sign_to_prediction_index_map.json\")\nindex_label = dict([(label_index[key], key) for key in label_index])\nprint(label_index)\ntrain[\"label\"] = train[\"sign\"].map(lambda sign: label_index[sign])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:14.013208Z","iopub.execute_input":"2023-04-25T11:11:14.013891Z","iopub.status.idle":"2023-04-25T11:11:14.063348Z","shell.execute_reply.started":"2023-04-25T11:11:14.013850Z","shell.execute_reply":"2023-04-25T11:11:14.062182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"sign\"].value_counts()\ntrain.num_frames.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:14.066457Z","iopub.execute_input":"2023-04-25T11:11:14.066945Z","iopub.status.idle":"2023-04-25T11:11:14.088136Z","shell.execute_reply.started":"2023-04-25T11:11:14.066905Z","shell.execute_reply":"2023-04-25T11:11:14.087204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization of Data","metadata":{}},{"cell_type":"code","source":"sample = pd.read_parquet(\"/kaggle/input/asl-signs/train_landmark_files/16069/100015657.parquet\")\nprint(f\"Sample shape = {sample.shape}\")\nsample.sample(12)\n\nsample_left_hand = sample[sample.type == \"left_hand\"]\nsample_right_hand = sample[sample.type == \"right_hand\"]\n\nprint(f\"Percentage of nulls in Left Hand data = {100*np.mean(sample_left_hand['x'].isnull()):.02f} %\")\nprint(f\"Percentage of nulls in Right Hand data = {100*np.mean(sample_right_hand['x'].isnull()):.02f} %\")\n\nedges = [(0,1),(1,2),(2,3),(3,4),(0,5),(0,17),(5,6),(6,7),(7,8),(5,9),(9,10),(10,11),(11,12),\n         (9,13),(13,14),(14,15),(15,16),(13,17),(17,18),(18,19),(19,20)]\n\ndef plot_frame(df, frame_id, ax):\n    df = df[df.frame == frame_id].sort_values(['landmark_index'])\n    x = list(df.x)\n    y = list(df.y)\n    \n    ax.scatter(df.x, df.y, color='black')\n    for i in range(len(x)):\n        ax.text(x[i], y[i], str(i))\n        \n    for edge in edges:\n        ax.plot([x[edge[0]], x[edge[1]]], [y[edge[0]], y[edge[1]]], color='lime')\n        ax.set_xlabel(f\"Frame no. {frame_id}\")\n        ax.set_xticks([])\n        ax.set_yticks([])\n        ax.set_xticklabels([])\n        ax.set_yticklabels([])\n\n    \ndef plot_frame_seq(df, frame_range, n_frames):\n    frames = np.linspace(frame_range[0],frame_range[1],n_frames, dtype = int, endpoint=True)\n    fig, ax = plt.subplots(n_frames, 1, figsize=(8,20))\n    for i in range(n_frames):\n        plot_frame(df, frames[i], ax[i])\n        \n    plt.show()\n\n    \nplot_frame_seq(sample_left_hand, (178,186), 5)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:14.091263Z","iopub.execute_input":"2023-04-25T11:11:14.092157Z","iopub.status.idle":"2023-04-25T11:11:15.084944Z","shell.execute_reply.started":"2023-04-25T11:11:14.092118Z","shell.execute_reply":"2023-04-25T11:11:15.083802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building\n","metadata":{}},{"cell_type":"code","source":"if cg.use_aggregation_dataset == False:\n    xs = []\n    ys = []\n    num_frames = np.zeros(len(train))\n    for i in tqdm(range(len(train))):\n        path = f\"{cg.data_path}{train.iloc[i].path}\"\n        data = load_relevant_data_subset_with_imputation(path)\n        ## Mean Aggregation\n        xs.append(np.mean(data, axis=0))\n        ys.append(train.iloc[i].label)\n        num_frames[i] = data.shape[0]\n        if cg.quick_experiment and i == 4999:\n            break\n    ## Save number of frames of each training sample for data analysis\n    train[\"num_frames\"] = num_frames\n    X = np.array(xs)\n    y = np.array(ys)\n    print(train[\"num_frames\"].describe())\n    train.to_csv(\"train.csv\", index=False)\nelse:\n    X = np.load(f\"{cg.aggregation_data_path}X.npy\")\n    y = np.load(f\"{cg.aggregation_data_path}y.npy\")\nprint(X.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:11:15.089934Z","iopub.execute_input":"2023-04-25T11:11:15.092548Z","iopub.status.idle":"2023-04-25T11:11:15.232606Z","shell.execute_reply.started":"2023-04-25T11:11:15.092502Z","shell.execute_reply":"2023-04-25T11:11:15.229877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model():\n    inputs = tf.keras.Input((543, 3), dtype=tf.float32)\n    vector = tf.keras.layers.Dense(128, activation=\"swish\")(inputs)\n    vector = tf.keras.layers.Dense(128, activation=\"swish\")(vector)\n    vector = tf.keras.layers.Dense(32, activation=\"swish\")(vector)\n    vector = tf.keras.layers.Dense(32, activation=\"swish\")(vector)\n    vector = tf.keras.layers.Dense(16, activation=\"swish\")(vector)\n    vector = tf.keras.layers.Dropout(0.1)(vector)\n    vector = tf.keras.layers.Flatten()(vector)\n    output = tf.keras.layers.Dense(256, activation=\"softmax\")(vector)\n    model = tf.keras.Model(inputs=inputs, outputs=output)\n    model.compile(\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n        metrics=[\n            \"accuracy\", \n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name=\"top-5-accuracy\"),\n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name=\"top-10-accuracy\")\n        ]\n    )\n    return model\n\n# def model():\n#     inputs = tf.keras.Input((543, 3), dtype=tf.float32)\n#     vector = tf.keras.layers.Dense(128, activation=\"swish\")(inputs)\n#     vector = tf.keras.layers.Dense(64, activation=\"swish\")(vector)\n#     vector = tf.keras.layers.Dense(32, activation=\"swish\")(vector)\n#     vector = tf.keras.layers.Dense(16, activation=\"swish\")(vector)\n#     vector = tf.keras.layers.Dropout(0.1)(vector)\n#     vector = tf.keras.layers.Flatten()(vector)\n#     output = tf.keras.layers.Dense(250, activation=\"softmax\")(vector)\n#     model = tf.keras.Model(inputs=inputs, outputs=output)\n#     model.compile(\n#         loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n#         metrics=[\n#             \"accuracy\", \n#             tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name=\"top-5-accuracy\"),\n#             tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name=\"top-10-accuracy\")\n#         ]\n#     )\n#     return model","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:25:48.058604Z","iopub.execute_input":"2023-04-25T11:25:48.059342Z","iopub.status.idle":"2023-04-25T11:25:48.071389Z","shell.execute_reply.started":"2023-04-25T11:25:48.059301Z","shell.execute_reply":"2023-04-25T11:25:48.070256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = \"model.h5\"\nif cg.is_training:\n    X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n    print(X_train.shape, y_train.shape, X_val.shape, y_val.shape)\n#     del X, y\n    gc.collect()\n    model = model()\n    callbacks = [\n        tf.keras.callbacks.ModelCheckpoint(model_name, save_best_only=True, restore_best_weights=True, monitor=\"val_accuracy\", mode=\"max\")\n    ]\n    model.fit(X_train, y_train, epochs=5, validation_data=(X_val, y_val), batch_size=128, callbacks=callbacks)\n    model.load_weights(model_name)\nelse:\n    model = tf.keras.models.load_model(f\"../input/sign-language-prediction-model/{model_name}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:25:53.196143Z","iopub.execute_input":"2023-04-25T11:25:53.196884Z","iopub.status.idle":"2023-04-25T11:26:41.535722Z","shell.execute_reply.started":"2023-04-25T11:25:53.196843Z","shell.execute_reply":"2023-04-25T11:26:41.534642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:19:42.349554Z","iopub.execute_input":"2023-04-25T11:19:42.350312Z","iopub.status.idle":"2023-04-25T11:19:42.375897Z","shell.execute_reply.started":"2023-04-25T11:19:42.350270Z","shell.execute_reply":"2023-04-25T11:19:42.375077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Live Model","metadata":{}},{"cell_type":"code","source":"def live_model(model):\n    inputs = tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")\n    x = tf.where(tf.math.is_nan(inputs), tf.zeros_like(inputs), inputs)\n    x = tf.reduce_mean(x, axis=0, keepdims=True)\n    x = model(x)\n    output = tf.keras.layers.Activation(activation=\"linear\", name=\"outputs\")(x)\n    inference_model = tf.keras.Model(inputs=inputs, outputs=output) \n    inference_model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=[\"accuracy\"])\n    return inference_model","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:12:21.742816Z","iopub.execute_input":"2023-04-25T11:12:21.743189Z","iopub.status.idle":"2023-04-25T11:12:21.756704Z","shell.execute_reply.started":"2023-04-25T11:12:21.743147Z","shell.execute_reply":"2023-04-25T11:12:21.755708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"live_model = live_model(model)\nlive_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:12:21.760384Z","iopub.execute_input":"2023-04-25T11:12:21.760816Z","iopub.status.idle":"2023-04-25T11:12:21.868040Z","shell.execute_reply.started":"2023-04-25T11:12:21.760773Z","shell.execute_reply":"2023-04-25T11:12:21.867239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making the Submission File","metadata":{}},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(live_model)\ntflite_model = converter.convert()\nmodel_path = \"model.tflite\"\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:12:21.869089Z","iopub.execute_input":"2023-04-25T11:12:21.869500Z","iopub.status.idle":"2023-04-25T11:12:25.804911Z","shell.execute_reply.started":"2023-04-25T11:12:21.869471Z","shell.execute_reply":"2023-04-25T11:12:25.803291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:12:25.806399Z","iopub.execute_input":"2023-04-25T11:12:25.807175Z","iopub.status.idle":"2023-04-25T11:12:27.283130Z","shell.execute_reply.started":"2023-04-25T11:12:25.807133Z","shell.execute_reply":"2023-04-25T11:12:27.281867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tflite-runtime","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:12:27.286211Z","iopub.execute_input":"2023-04-25T11:12:27.286646Z","iopub.status.idle":"2023-04-25T11:12:37.199846Z","shell.execute_reply.started":"2023-04-25T11:12:27.286575Z","shell.execute_reply":"2023-04-25T11:12:37.198539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tflite_runtime.interpreter as tflite\ninterpreter = tflite.Interpreter(model_path)\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\nfor i in tqdm(range(10000)):\n    frames = load_relevant_data_subset(f'/kaggle/input/asl-signs/{train.iloc[i].path}')\n    output = prediction_fn(inputs=frames)\n    sign = np.argmax(output[\"outputs\"])\n    if i % 100 == 0:\n        print(f\"Predicted Label: {index_label[sign]}, Actual Label: {train.iloc[i].sign}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:12:37.203765Z","iopub.execute_input":"2023-04-25T11:12:37.204106Z","iopub.status.idle":"2023-04-25T11:14:14.710890Z","shell.execute_reply.started":"2023-04-25T11:12:37.204070Z","shell.execute_reply":"2023-04-25T11:14:14.709760Z"},"trusted":true},"execution_count":null,"outputs":[]}]}