{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Sign Language Recognition Challenge\nThe goal of this competition is to classify isolated American Sign Language( ASL ) signs.\n\nThe landmarks were extracted from raw videos with the Mediapipe holistic model and are asked to predict the sign from this data.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\n\nplt.style.use(\"seaborn-colorblind\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:21.952397Z","iopub.execute_input":"2023-03-24T19:59:21.952813Z","iopub.status.idle":"2023-03-24T19:59:23.030477Z","shell.execute_reply.started":"2023-03-24T19:59:21.952776Z","shell.execute_reply":"2023-03-24T19:59:23.029348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install nb_black for autoformatting\n!pip install nb_black --quiet\n%load_ext lab_black","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:23.032656Z","iopub.execute_input":"2023-03-24T19:59:23.033984Z","iopub.status.idle":"2023-03-24T19:59:39.591994Z","shell.execute_reply.started":"2023-03-24T19:59:23.033929Z","shell.execute_reply":"2023-03-24T19:59:39.590904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data EDA","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"../input/asl-signs\"\ntrain = pd.read_csv(f\"{BASE_DIR}/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:39.593670Z","iopub.execute_input":"2023-03-24T19:59:39.594435Z","iopub.status.idle":"2023-03-24T19:59:39.814875Z","shell.execute_reply.started":"2023-03-24T19:59:39.594393Z","shell.execute_reply":"2023-03-24T19:59:39.813414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train.csv has the path to each parquet file, the participant id, sequence_id and sign\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:39.817450Z","iopub.execute_input":"2023-03-24T19:59:39.817848Z","iopub.status.idle":"2023-03-24T19:59:39.855859Z","shell.execute_reply.started":"2023-03-24T19:59:39.817793Z","shell.execute_reply":"2023-03-24T19:59:39.854401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What signs are we trying to predict?\n* 250 Unique Signs\n* Ranging from 299-415 Examples of Each","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 8))\ntrain[\"sign\"].value_counts().head(50).sort_values(ascending=True).plot(\n    kind=\"barh\", figsize=(8, 8), title=\"Top 50 signs in Training data set\"\n)\nax.set_xlabel(\"Number of Training Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:39.857756Z","iopub.execute_input":"2023-03-24T19:59:39.858135Z","iopub.status.idle":"2023-03-24T19:59:40.819457Z","shell.execute_reply.started":"2023-03-24T19:59:39.858098Z","shell.execute_reply":"2023-03-24T19:59:40.818487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 8))\ntrain[\"sign\"].value_counts().tail(50).sort_values(ascending=True).plot(\n    kind=\"barh\", figsize=(8, 8), title=\"Bottom 50 signs in Training data set\"\n)\nax.set_xlabel(\"Number of Training Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:40.820676Z","iopub.execute_input":"2023-03-24T19:59:40.821240Z","iopub.status.idle":"2023-03-24T19:59:41.685266Z","shell.execute_reply.started":"2023-03-24T19:59:40.821205Z","shell.execute_reply":"2023-03-24T19:59:41.683853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Praquet Landmark Data\n\n- Each parquet file is in the path:\n    - train_landmark_files/[participant_id]/[sequence_id].parquet\n- The parquet's assosiated sign can be found in train.csv","metadata":{}},{"cell_type":"markdown","source":"## Pull an example parquet file data..\n\nwe pull an example landmark file for the sign \"listen\"","metadata":{}},{"cell_type":"code","source":"example_fn = train.query('sign==\"listen\"')[\"path\"].values[0]\n\nexample_landmark = pd.read_parquet(f\"{BASE_DIR}/{example_fn}\")\nexample_landmark.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:41.686640Z","iopub.execute_input":"2023-03-24T19:59:41.687027Z","iopub.status.idle":"2023-03-24T19:59:41.831760Z","shell.execute_reply.started":"2023-03-24T19:59:41.686960Z","shell.execute_reply":"2023-03-24T19:59:41.830858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_frames = example_landmark[\"frame\"].nunique()\nunique_types = example_landmark[\"type\"].nunique()\nprint(\n    f\"This file has {unique_frames} - unique frames and {unique_types} - unique types\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:41.834534Z","iopub.execute_input":"2023-03-24T19:59:41.835001Z","iopub.status.idle":"2023-03-24T19:59:41.848791Z","shell.execute_reply.started":"2023-03-24T19:59:41.834937Z","shell.execute_reply":"2023-03-24T19:59:41.847315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's compare for a bunch of parquet files what type of data we have.\n\nwe notice the number of frames is not consistant alomst every file has 4 types of landmarks face, left hand, pose, right hand.","metadata":{}},{"cell_type":"code","source":"listen_files = train.query('sign == \"listen\"')[\"path\"].values\nfor i, f in enumerate(listen_files):\n    example_landmark = pd.read_parquet(f\"{BASE_DIR}/{f}\")\n    unique_frames = example_landmark[\"frame\"].nunique()\n    unique_types = example_landmark[\"type\"].nunique()\n    types_in_video = example_landmark[\"type\"].unique()\n    print(\n        f\"The file has {unique_frames} unique frames and {unique_types} unique types: {types_in_video}\"\n    )\n    if i == 20:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:41.850862Z","iopub.execute_input":"2023-03-24T19:59:41.851812Z","iopub.status.idle":"2023-03-24T19:59:42.513202Z","shell.execute_reply.started":"2023-03-24T19:59:41.851763Z","shell.execute_reply":"2023-03-24T19:59:42.511724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Metadata for Training Dataset","metadata":{}},{"cell_type":"code","source":"N_PARQUETS_TO_READ = 1_000  # So we don't have to load all 95k\n\ncombined_meta = {}\nfor i, d in tqdm(train.iterrows(), total=len(train)):\n    file_path = d[\"path\"]\n    example_landmark = pd.read_parquet(f\"{BASE_DIR}/{file_path}\")\n    # Get the number of landmarks with x,y,z data\n    meta = (\n        example_landmark.dropna(subset=[\"x\", \"y\", \"z\"])[\"type\"].value_counts().to_dict()\n    )\n    meta[\"frames\"] = example_landmark[\"frame\"].nunique()\n    xyz_meta = (\n        example_landmark.agg(\n            {\n                \"x\": [\"min\", \"max\", \"mean\"],\n                \"y\": [\"min\", \"max\", \"mean\"],\n                \"z\": [\"min\", \"max\", \"mean\"],\n            }\n        )\n        .unstack()\n        .to_dict()\n    )\n    for key in xyz_meta.keys():\n        new_key = key[0] + \"_\" + key[1]\n        meta[new_key] = xyz_meta[key]\n    combined_meta[file_path] = meta\n    if i >= N_PARQUETS_TO_READ:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:59:42.517879Z","iopub.execute_input":"2023-03-24T19:59:42.518245Z","iopub.status.idle":"2023-03-24T20:00:18.888543Z","shell.execute_reply.started":"2023-03-24T19:59:42.518213Z","shell.execute_reply":"2023-03-24T20:00:18.887214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_with_meta = train.merge(\n    pd.DataFrame(combined_meta).T.reset_index().rename(columns={\"index\": \"path\"}),\n    how=\"left\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:18.890296Z","iopub.execute_input":"2023-03-24T20:00:18.891447Z","iopub.status.idle":"2023-03-24T20:00:19.006261Z","shell.execute_reply.started":"2023-03-24T20:00:18.891396Z","shell.execute_reply":"2023-03-24T20:00:19.004905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_with_meta[[\"face\", \"pose\", \"left_hand\", \"right_hand\"]].sum().sort_values().plot(\n    kind=\"barh\", title=\"Sum of Rows by Landmark Type\"\n)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:19.007642Z","iopub.execute_input":"2023-03-24T20:00:19.007971Z","iopub.status.idle":"2023-03-24T20:00:19.225229Z","shell.execute_reply.started":"2023-03-24T20:00:19.007940Z","shell.execute_reply":"2023-03-24T20:00:19.223950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking to see of the number of landmarks for this type is zero\n(\n    train_with_meta.query(\"index < 1000\").fillna(0)[\n        [\"face\", \"pose\", \"left_hand\", \"right_hand\"]\n    ]\n    > 0\n).mean().plot(kind=\"barh\", title=\"percent of Frames/keypoints with Data\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:19.226657Z","iopub.execute_input":"2023-03-24T20:00:19.227016Z","iopub.status.idle":"2023-03-24T20:00:19.456827Z","shell.execute_reply.started":"2023-03-24T20:00:19.226954Z","shell.execute_reply":"2023-03-24T20:00:19.455539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check out one example","metadata":{}},{"cell_type":"code","source":"example_fn = train_with_meta.dropna().query('sign==\"shhh\"')[\"path\"].values[0]\n\nexample_landmark = pd.read_parquet(f\"{BASE_DIR}/{example_fn}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:19.458580Z","iopub.execute_input":"2023-03-24T20:00:19.459068Z","iopub.status.idle":"2023-03-24T20:00:19.502268Z","shell.execute_reply.started":"2023-03-24T20:00:19.459019Z","shell.execute_reply":"2023-03-24T20:00:19.500855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_landmark.query(\"frame == 25\")[\"type\"].value_counts()  # Middle of the video","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:19.503743Z","iopub.execute_input":"2023-03-24T20:00:19.504120Z","iopub.status.idle":"2023-03-24T20:00:19.522126Z","shell.execute_reply.started":"2023-03-24T20:00:19.504084Z","shell.execute_reply":"2023-03-24T20:00:19.521015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3D plot of Landmarks from \"shhh\" example","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\nexample_frame = example_landmark.query(\"frame == 25\")\npx.scatter_3d(example_frame, x=\"x\", y=\"y\", z=\"z\", color=\"type\")","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:19.523340Z","iopub.execute_input":"2023-03-24T20:00:19.523650Z","iopub.status.idle":"2023-03-24T20:00:23.288045Z","shell.execute_reply.started":"2023-03-24T20:00:19.523621Z","shell.execute_reply":"2023-03-24T20:00:23.286784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation\nThe evaluation metric for this contest is simple classification accuracy","metadata":{}},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\n\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = [\"x\", \"y\", \"z\"]\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:23.289412Z","iopub.execute_input":"2023-03-24T20:00:23.289733Z","iopub.status.idle":"2023-03-24T20:00:23.301567Z","shell.execute_reply.started":"2023-03-24T20:00:23.289702Z","shell.execute_reply":"2023-03-24T20:00:23.300306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tflite_runtime.interpreter as tflite\n# interpreter = tflite.Interpreter(model_path)\n\n# found_signatures = list(interpreter.get_signature_list().keys())\n\n# if REQUIRED_SIGNATURE not in found_signatures:\n#     raise KernelEvalException('Required input signature not found.')\n\n# prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n# output = prediction_fn(inputs=frames)\n# sign = np.argmax(output[\"outputs\"])","metadata":{"execution":{"iopub.status.busy":"2023-03-24T20:00:23.303659Z","iopub.execute_input":"2023-03-24T20:00:23.307141Z","iopub.status.idle":"2023-03-24T20:00:23.540911Z","shell.execute_reply.started":"2023-03-24T20:00:23.307087Z","shell.execute_reply":"2023-03-24T20:00:23.539211Z"},"trusted":true},"execution_count":null,"outputs":[]}]}