{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Libraries\n\nimport json\nimport math\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm # It shows nice progress bars\n\n# Input data files are available in the read-only \"../input/\" directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-01T00:55:41.042560Z","iopub.execute_input":"2023-03-01T00:55:41.043367Z","iopub.status.idle":"2023-03-01T00:55:52.347089Z","shell.execute_reply.started":"2023-03-01T00:55:41.043313Z","shell.execute_reply":"2023-03-01T00:55:52.345931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Load and show training data","metadata":{}},{"cell_type":"markdown","source":"## Set base directories","metadata":{}},{"cell_type":"code","source":"asl_signs_dir = '../input/asl-signs/'","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.349856Z","iopub.execute_input":"2023-03-01T00:55:52.351233Z","iopub.status.idle":"2023-03-01T00:55:52.357781Z","shell.execute_reply.started":"2023-03-01T00:55:52.351158Z","shell.execute_reply":"2023-03-01T00:55:52.356199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show contents of train.csv","metadata":{}},{"cell_type":"code","source":"train_file = 'train.csv'\ntrain_df = pd.read_csv(os.path.join(asl_signs_dir, train_file), encoding='utf8')\n\nprint(f'Samples, columns: {train_df.shape}')\nprint(train_df.dtypes)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.360138Z","iopub.execute_input":"2023-03-01T00:55:52.360719Z","iopub.status.idle":"2023-03-01T00:55:52.629352Z","shell.execute_reply.started":"2023-03-01T00:55:52.360662Z","shell.execute_reply":"2023-03-01T00:55:52.628002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show the contents of a random file in path column of train.csv","metadata":{}},{"cell_type":"code","source":"random_sample = train_df.sample()\nrandom_sample_path = random_sample['path'].values[0]\nprint(random_sample_path)\n\nparquet_df = pd.read_parquet(os.path.join(asl_signs_dir, random_sample_path))\nprint(parquet_df.dtypes)\nparquet_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.632056Z","iopub.execute_input":"2023-03-01T00:55:52.632411Z","iopub.status.idle":"2023-03-01T00:55:52.793494Z","shell.execute_reply.started":"2023-03-01T00:55:52.632377Z","shell.execute_reply":"2023-03-01T00:55:52.792510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, for a given sign done by a participant (in a given sequence), we have a set of landmarks that describe the sign in a parquet file. It would be nice to see the landmarks for a random sign.","metadata":{}},{"cell_type":"markdown","source":"## Check if the data is clean","metadata":{}},{"cell_type":"code","source":"print(f\"# null values: {parquet_df.isnull().sum().sum()}\")\nprint(f\"# rows: {len(parquet_df)}, # rows without duplicates: \"\n      f\"{len(parquet_df.drop_duplicates())}\")\nparquet_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.794893Z","iopub.execute_input":"2023-03-01T00:55:52.795994Z","iopub.status.idle":"2023-03-01T00:55:52.849880Z","shell.execute_reply.started":"2023-03-01T00:55:52.795955Z","shell.execute_reply":"2023-03-01T00:55:52.848391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, there are null values, no duplicates, and the normalized coordinates are not in the range [0, 1]. We can just remove all weird values.","metadata":{}},{"cell_type":"code","source":"# Remove null\nparquet_df_clean = parquet_df.dropna()\n\n# Remove rows where x or y are not in [0, 1]. Ignore z for now\nparquet_df_clean = parquet_df_clean[(parquet_df_clean.x >= 0) & \n                                    (parquet_df_clean.x <= 1) &\n                                    (parquet_df_clean.y >= 0) & \n                                    (parquet_df_clean.y <= 1)]\n\n# Note, we could remove the entire landmark of a frame if one of its\n# coordinates is bad (a whole face or hand), and not just one point,\n# but I am lazy at the moment","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.852067Z","iopub.execute_input":"2023-03-01T00:55:52.853554Z","iopub.status.idle":"2023-03-01T00:55:52.868348Z","shell.execute_reply.started":"2023-03-01T00:55:52.853470Z","shell.execute_reply":"2023-03-01T00:55:52.866795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check cleanliness again\nprint(f\"# null values: {parquet_df_clean.isnull().sum().sum()}\")\nprint(f\"# rows: {len(parquet_df_clean)}, # rows without duplicates: \"\n      f\"{len(parquet_df_clean.drop_duplicates())}\")\nparquet_df_clean.describe()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.870205Z","iopub.execute_input":"2023-03-01T00:55:52.871321Z","iopub.status.idle":"2023-03-01T00:55:52.930385Z","shell.execute_reply.started":"2023-03-01T00:55:52.871262Z","shell.execute_reply":"2023-03-01T00:55:52.929056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display landmarks for a random sign (listed in a parquet file), just in 2D for now as depth can be tricky","metadata":{}},{"cell_type":"markdown","source":"Let's see which landmark types and frame indexes are in the randon parquet we chose and cleaned","metadata":{}},{"cell_type":"code","source":"# Print landmark types per frame in a parquet\nfor frame_id in parquet_df_clean.frame.unique().tolist():\n    list_types = []\n    print(f\"Frame: {frame_id}\")\n    \n    for landmark_type in parquet_df_clean[\n        parquet_df_clean.frame == frame_id].type.unique().tolist():\n        list_types.append(landmark_type)\n        \n    print(list_types)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.934363Z","iopub.execute_input":"2023-03-01T00:55:52.935929Z","iopub.status.idle":"2023-03-01T00:55:52.956363Z","shell.execute_reply.started":"2023-03-01T00:55:52.935867Z","shell.execute_reply":"2023-03-01T00:55:52.954901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define some global values for the plots\nedges = [(0,1),(1,2),(2,3),(3,4),(0,5),(0,17),(5,6),\n         (6,7),(7,8),(5,9),(9,10),(10,11),(11,12),\n         (9,13),(13,14),(14,15),(15,16),(13,17),\n         (17,18),(18,19),(19,20)]\n\ncolors = ['blue', 'green', 'orange', 'brown', 'purple']\n\n# Plot one specific type of landmark\ndef plot_frame_landmark(parquet_df, frame_id, landmark_type, axes):\n    parquet_df_sorted_filtered = parquet_df[\n            (parquet_df.frame == frame_id) &\n            (parquet_df.type == landmark_type)\n        ].sort_values(['landmark_index'])\n    \n    x = list(parquet_df_sorted_filtered.x)\n    y = list(parquet_df_sorted_filtered.y)\n\n    axes.scatter(x, y, color='blue')\n    \n    for i in range(len(x)):\n        axes.text(x[i], y[i], str(i))\n        \n    axes.set_xlabel(f\"Frame {frame_id} - {landmark_type}\")\n    \n    # Draw edges if it is a hand, for better visualization\n    # Nice code from:\n    # https://www.kaggle.com/code/mayukh18/sign-language-eda-visualization/notebook\n    \n    if 'hand' in landmark_type:\n        for edge in edges:\n            axes.plot([x[edge[0]], x[edge[1]]], \n                      [y[edge[0]], y[edge[1]]], \n                      color='red')\n            \n# Choose a frame and landmark type to plot\nframe_id_show = 34\nlandmark_type_show = 'right_hand'\n\n# Make sure the frame - landmark type combo exists\nif frame_id_show in parquet_df_clean.frame.unique().tolist():\n    if landmark_type_show in parquet_df_clean[\n        parquet_df_clean.frame == frame_id_show].type.unique().tolist():\n            _, axes = plt.subplots(1, 1, figsize=(5, 5))\n            plot_frame_landmark(parquet_df_clean, frame_id_show, landmark_type_show, axes)      \n            plt.gca().invert_yaxis() # Invert the y axis to see better\n            plt.show()\n    else:\n        print(\"The landmark you chose does not exist in the frame of the parquet.\")\nelse:\n    print(\"The frame you chose does not exist in the parquet.\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.958463Z","iopub.execute_input":"2023-03-01T00:55:52.959059Z","iopub.status.idle":"2023-03-01T00:55:52.977367Z","shell.execute_reply.started":"2023-03-01T00:55:52.959008Z","shell.execute_reply":"2023-03-01T00:55:52.975873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the whole frame with all landmarks\ndef plot_frame(parquet_df, frame_id, axes):\n    \n    color_id = 0\n    \n    for landmark_type in parquet_df[\n        parquet_df.frame == frame_id].type.unique().tolist():\n        \n        parquet_df_sorted_filtered = parquet_df[\n                (parquet_df.frame == frame_id) &\n                (parquet_df.type == landmark_type)\n            ].sort_values(['landmark_index'])\n\n        x = list(parquet_df_sorted_filtered.x)\n        y = list(parquet_df_sorted_filtered.y)\n\n        axes.scatter(x, y, color=colors[color_id])\n        color_id += 1\n\n        for i in range(len(x)):\n            axes.text(x[i], y[i], str(i))\n\n        # Add edges for a hand\n        if 'hand' in landmark_type:\n            for edge in edges:\n                axes.plot([x[edge[0]], x[edge[1]]], \n                          [y[edge[0]], y[edge[1]]], \n                          color='red')\n        \n    axes.set_xlabel(f\"Frame {frame_id} - {landmark_type}\")\n    \n# Choose a frame to plot\nframe_id_show = 34\n\n# Make sure the frame exists, to avoid errors\nif frame_id_show in parquet_df_clean.frame.unique().tolist():\n    _, axes = plt.subplots(1, 1, figsize=(15, 15))\n    plot_frame(parquet_df_clean, frame_id_show, axes)      \n    plt.gca().invert_yaxis() # Invert the y axis to see better\n    plt.show()\nelse:\n    print(\"The frame you chose does not exist in the parquet.\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:52.982920Z","iopub.execute_input":"2023-03-01T00:55:52.983602Z","iopub.status.idle":"2023-03-01T00:55:55.109820Z","shell.execute_reply.started":"2023-03-01T00:55:52.983547Z","shell.execute_reply":"2023-03-01T00:55:55.108727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we can plot a sequence of frames that mean a sign","metadata":{}},{"cell_type":"code","source":"def plot_frames_parquet(parquet_df):\n    frames_in_parquet = np.sort(parquet_df.frame.unique()).tolist()\n    n_frames = len(frames_in_parquet)\n    n_cols = 3\n    n_rows = math.floor(n_frames / n_cols) + 1   \n    \n    _, axes = plt.subplots(n_rows, n_cols, figsize=(5 * n_cols, 5 * n_rows))\n      \n    for i in range(n_frames):\n        n_row = math.floor(i / n_cols)\n        n_col = i % n_cols\n        plot_frame(parquet_df, frames_in_parquet[i], axes[n_row][n_col])\n        axes[n_row][n_col].invert_yaxis() # It looks better inverted\n        \nprint(random_sample.sign.values[0]) # Sign value\n# plot_frames_parquet(parquet_df_clean)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:55.111042Z","iopub.execute_input":"2023-03-01T00:55:55.111385Z","iopub.status.idle":"2023-03-01T00:55:55.123322Z","shell.execute_reply.started":"2023-03-01T00:55:55.111350Z","shell.execute_reply":"2023-03-01T00:55:55.121576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Prepare a dataset to train a neural network","metadata":{}},{"cell_type":"code","source":"# Check that sequence id identifies each row in train dataset with no duplicates\n\nprint(f\"# unique sequence ids: {len(np.unique(train_df.sequence_id.values))}\\n\"\n      f\"# rows train: {len(train_df)}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:55.125381Z","iopub.execute_input":"2023-03-01T00:55:55.126156Z","iopub.status.idle":"2023-03-01T00:55:55.140449Z","shell.execute_reply.started":"2023-03-01T00:55:55.126101Z","shell.execute_reply":"2023-03-01T00:55:55.139040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The instructions indicate that the model must take one or more landmark frames as an input and return a float vector (the predicted probabilities of each sign class) as the output and data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns)).\n\nSo, the input has to contain an array of landmark coordinates in a frame, with ROWS_PER_FRAME rows per frame (each row is a landmark coord), and 3 columns: x, y, and z. That's what I understood up to now. ","metadata":{}},{"cell_type":"code","source":"# Function to load data taken directly from:\n# https://www.kaggle.com/competitions/asl-signs/overview/evaluation\n\nROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:55.142369Z","iopub.execute_input":"2023-03-01T00:55:55.142854Z","iopub.status.idle":"2023-03-01T00:55:55.152958Z","shell.execute_reply.started":"2023-03-01T00:55:55.142807Z","shell.execute_reply":"2023-03-01T00:55:55.151560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the mapping of signs to codes\ndict_signs_name = \"sign_to_prediction_index_map.json\"\ndict_signs_path = os.path.join(asl_signs_dir, dict_signs_name)\ndict_sign_to_code = None\nwith open(dict_signs_path, \"r\") as f:\n    dict_sign_to_code = json.load(f)\n\nprint(dict_sign_to_code)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:55.154178Z","iopub.execute_input":"2023-03-01T00:55:55.154561Z","iopub.status.idle":"2023-03-01T00:55:55.172239Z","shell.execute_reply.started":"2023-03-01T00:55:55.154508Z","shell.execute_reply":"2023-03-01T00:55:55.170855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xs = None\nys = []\n\n# Use only the first 250, because there are just too many rows\nfor row in tqdm(train_df.head(250).itertuples()):\n    full_path = os.path.join(asl_signs_dir, row.path)    \n    \n    # We need to clean and do some preprocessing, but as the model\n    # has to do all that with the test data too, I will include\n    # the preprocessing in the tensorflow model    \n    loaded_data = load_relevant_data_subset(full_path)\n    \n    xs = loaded_data if xs is None else np.concatenate((xs, loaded_data), axis=0)\n    \n    for i in range(len(loaded_data)):\n        ys.append(dict_sign_to_code[row.sign])\n    \nprint(xs.shape)\nys = np.array(ys)\nprint(ys.shape)\nprint(xs[0])\nprint(ys)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T00:55:55.174315Z","iopub.execute_input":"2023-03-01T00:55:55.174793Z","iopub.status.idle":"2023-03-01T00:56:02.551656Z","shell.execute_reply.started":"2023-03-01T00:55:55.174745Z","shell.execute_reply":"2023-03-01T00:56:02.550333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Try with a random model I got from:\n# https://www.kaggle.com/code/lonnieqin/isolated-sign-language-recognition-with-dnn\ndef get_model():\n    inputs = tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")\n    vector = tf.where(tf.math.is_nan(inputs), tf.zeros_like(inputs), inputs)\n    vector = tf.keras.layers.Dense(128, activation=\"relu\")(vector)\n    vector = tf.keras.layers.Dense(64, activation=\"relu\")(vector)\n    vector = tf.keras.layers.Dense(32, activation=\"relu\")(vector)\n    vector = tf.keras.layers.Dense(16, activation=\"relu\")(vector)\n    vector = tf.keras.layers.Flatten()(vector)\n    vector = tf.keras.layers.Dense(250, activation=\"softmax\")(vector)\n    output = tf.keras.layers.Activation(activation=\"linear\", name=\"outputs\")(vector)\n    model = tf.keras.Model(inputs=inputs, outputs=output)\n    learning_rate = 1e-3\n    optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate)\n    model.compile(\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n        metrics=[\n            \"accuracy\", \n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10)\n        ],\n        optimizer=optimizer\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:07:31.393453Z","iopub.execute_input":"2023-03-01T01:07:31.393926Z","iopub.status.idle":"2023-03-01T01:07:31.405830Z","shell.execute_reply.started":"2023-03-01T01:07:31.393883Z","shell.execute_reply":"2023-03-01T01:07:31.404375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(xs, ys, test_size=0.2, random_state=42)\nprint(X_train.shape, y_train.shape, X_val.shape, y_val.shape)\n\n# del xs, ys\n# gc.collect()\n\nmodel = get_model()\ncallbacks = [tf.keras.callbacks.ModelCheckpoint(\"model.h5\")]\nmodel.fit(X_train, y_train, epochs=3, validation_data=(X_val, y_val), batch_size=128, callbacks=callbacks)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:07:38.218593Z","iopub.execute_input":"2023-03-01T01:07:38.219090Z","iopub.status.idle":"2023-03-01T01:08:20.902757Z","shell.execute_reply.started":"2023-03-01T01:07:38.219046Z","shell.execute_reply":"2023-03-01T01:08:20.901272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict(xs[0:5])\nprint(np.argmax(prediction[0]))\nprint(prediction[0])","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:08:20.905534Z","iopub.execute_input":"2023-03-01T01:08:20.905919Z","iopub.status.idle":"2023-03-01T01:08:21.141176Z","shell.execute_reply.started":"2023-03-01T01:08:20.905883Z","shell.execute_reply":"2023-03-01T01:08:21.139758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(model)\ntflite_model = converter.convert()\nmodel_path = \"model.tflite\"\n# Save the model.\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:08:34.588355Z","iopub.execute_input":"2023-03-01T01:08:34.588779Z","iopub.status.idle":"2023-03-01T01:08:37.055000Z","shell.execute_reply.started":"2023-03-01T01:08:34.588741Z","shell.execute_reply":"2023-03-01T01:08:37.053862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:08:39.268124Z","iopub.execute_input":"2023-03-01T01:08:39.268553Z","iopub.status.idle":"2023-03-01T01:08:40.935423Z","shell.execute_reply.started":"2023-03-01T01:08:39.268508Z","shell.execute_reply":"2023-03-01T01:08:40.933975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tflite-runtime","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:05:37.464154Z","iopub.execute_input":"2023-03-01T01:05:37.464667Z","iopub.status.idle":"2023-03-01T01:05:51.063945Z","shell.execute_reply.started":"2023-03-01T01:05:37.464625Z","shell.execute_reply":"2023-03-01T01:05:51.062533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tflite_runtime.interpreter as tflite\ninterpreter = tflite.Interpreter(model_path)\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:08:43.569625Z","iopub.execute_input":"2023-03-01T01:08:43.570057Z","iopub.status.idle":"2023-03-01T01:08:43.578790Z","shell.execute_reply.started":"2023-03-01T01:08:43.570018Z","shell.execute_reply":"2023-03-01T01:08:43.577397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = prediction_fn(inputs=xs[0:5])\nsign = np.argmax(output[\"outputs\"][0])\nprint(sign)\n#print(f\"Predicted label: {sign}, Actual Label: {ys[]}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T01:11:44.263582Z","iopub.execute_input":"2023-03-01T01:11:44.264030Z","iopub.status.idle":"2023-03-01T01:11:44.283376Z","shell.execute_reply.started":"2023-03-01T01:11:44.263993Z","shell.execute_reply":"2023-03-01T01:11:44.281706Z"},"trusted":true},"execution_count":null,"outputs":[]}]}