{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Sign Language Recognition Challenge\n- The goal is to classify isolated American sign language hand gestures (Multi classification).\n- The data are landmarks that were extracted from raw videos with the MediaPipe holistic model. Not all of the frames necessarily had visible hands or hands that could be detected by the model.\n-  Not all of the frames necessarily had visible hands or hands that could be detected by the model.\n\n \n### Trello: https://trello.com/b/dMkk0x4c/googlesign-language","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:08:19.851597Z","iopub.execute_input":"2023-02-26T21:08:19.852023Z","iopub.status.idle":"2023-02-26T21:08:19.879471Z","shell.execute_reply.started":"2023-02-26T21:08:19.851983Z","shell.execute_reply":"2023-02-26T21:08:19.878541Z"}}},{"cell_type":"code","source":"\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport json\nimport plotly.graph_objects as go\nimport plotly.io as pio\npio.templates.default = \"simple_white\"\n\n\nplt.style.use('seaborn-colorblind')\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-21T08:53:47.368910Z","iopub.execute_input":"2023-03-21T08:53:47.370177Z","iopub.status.idle":"2023-03-21T08:54:00.448583Z","shell.execute_reply.started":"2023-03-21T08:53:47.370131Z","shell.execute_reply":"2023-03-21T08:54:00.446974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"INPUT_DIR = \"/kaggle/input/asl-signs/\"\ntrain = pd.read_csv(f\"{INPUT_DIR}train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T08:54:00.451192Z","iopub.execute_input":"2023-03-21T08:54:00.451962Z","iopub.status.idle":"2023-03-21T08:54:00.768321Z","shell.execute_reply.started":"2023-03-21T08:54:00.451916Z","shell.execute_reply":"2023-03-21T08:54:00.767166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.query(\"sign == 'blow'\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T08:54:00.770550Z","iopub.execute_input":"2023-03-21T08:54:00.771475Z","iopub.status.idle":"2023-03-21T08:54:00.824590Z","shell.execute_reply.started":"2023-03-21T08:54:00.771415Z","shell.execute_reply":"2023-03-21T08:54:00.822482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many data points we have?\nlen(train)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T08:54:00.827233Z","iopub.execute_input":"2023-03-21T08:54:00.827829Z","iopub.status.idle":"2023-03-21T08:54:00.835400Z","shell.execute_reply.started":"2023-03-21T08:54:00.827753Z","shell.execute_reply":"2023-03-21T08:54:00.834146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What are the signs?\n- We have 250 unique signs\n- Ranging between 299 to 415 each\n\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 8))\n\ntrain.sign.value_counts().head(50).sort_values(ascending=True).plot(\n    kind=\"barh\", ax=ax, figsize=(10, 8), title=\"Top 50 signs\")\n\nax.set_xlabel(\"Number of Traiining Examples\")\nax.set_ylabel(\"The Sign Label\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:17.629913Z","iopub.execute_input":"2023-03-21T04:53:17.630650Z","iopub.status.idle":"2023-03-21T04:53:18.476744Z","shell.execute_reply.started":"2023-03-21T04:53:17.630602Z","shell.execute_reply":"2023-03-21T04:53:18.475400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 8))\n\ntrain.sign.value_counts().tail(50).sort_values(ascending=True).plot(\n    kind=\"barh\", ax=ax, figsize=(10, 8), title=\"Bottom 50 signs\")\n\nax.set_xlabel(\"Number of Traiining Examples\")\nax.set_ylabel(\"The Sign Label\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:18.478101Z","iopub.execute_input":"2023-03-21T04:53:18.478435Z","iopub.status.idle":"2023-03-21T04:53:19.236345Z","shell.execute_reply.started":"2023-03-21T04:53:18.478401Z","shell.execute_reply":"2023-03-21T04:53:19.235092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prequet Landmark Data\n### Observations:\n- The path to each parquet file is stored in train_landmark_files/[participant_id]/[sequence_id].parquet\n- In every parquet file we have multiple frames.\n- for every frame we have all the following types listed in the dataframe: ['face', 'left_hand', 'pose', 'right_hand'], but some types have NAN values for the x, y, z coordinates, marking that the type doesn't exist in the frame.\n- Eache type in a `frame` has diffirent indicies (the pose estimation points) example: right_hand landmark_index 1, right_hand landmark_index 2, face landmark_index 1, face landmark_index 1, etc.\n\n- Every type has fixed number of points: face: 468, left_hand: 21 , right_hand: 21 , pose: around 33 (the whole body not always captured in a given frame)\n","metadata":{}},{"cell_type":"code","source":"# Select and example poarquet file from (here we are selecting one of the for the word listen)\nexample_fn = train.query(\"sign== 'shhh'\")[\"path\"].values[0]\nexample_landmark = pd.read_parquet(f\"{INPUT_DIR}/{example_fn}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.238018Z","iopub.execute_input":"2023-03-21T04:53:19.238851Z","iopub.status.idle":"2023-03-21T04:53:19.397910Z","shell.execute_reply.started":"2023-03-21T04:53:19.238809Z","shell.execute_reply":"2023-03-21T04:53:19.396524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(example_landmark )","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.399778Z","iopub.execute_input":"2023-03-21T04:53:19.400127Z","iopub.status.idle":"2023-03-21T04:53:19.406733Z","shell.execute_reply.started":"2023-03-21T04:53:19.400092Z","shell.execute_reply":"2023-03-21T04:53:19.405787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_landmark[\"type\"].unique()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.408112Z","iopub.execute_input":"2023-03-21T04:53:19.409054Z","iopub.status.idle":"2023-03-21T04:53:19.424921Z","shell.execute_reply.started":"2023-03-21T04:53:19.409014Z","shell.execute_reply":"2023-03-21T04:53:19.423991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nunique_frames = example_landmark[\"frame\"].nunique()\nnunique_types = example_landmark[\"type\"].nunique()\nprint(f\"There is {nunique_frames} and {nunique_types} in example_landmark\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.428265Z","iopub.execute_input":"2023-03-21T04:53:19.429463Z","iopub.status.idle":"2023-03-21T04:53:19.443088Z","shell.execute_reply.started":"2023-03-21T04:53:19.429424Z","shell.execute_reply":"2023-03-21T04:53:19.441888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization\n\nThe visualization part is adapted from the following notebook: https://www.kaggle.com/code/josephzahar/interactive-3d-animated-visualization-of-asl\n\nMake sure to give a thumbs up! :)\n\n## Landmark & body keypoints\nface landmarks: https://github.com/tensorflow/tfjs-models/blob/838611c02f51159afdd77469ce67f0e26b7bbb23/face-landmarks-detection/src/mediapipe-facemesh/keypoints.ts\n\n\n### The Pose Keypoints\n![image](https://mediapipe.dev/images/mobile/pose_tracking_full_body_landmarks.png)\n\n### The Hands Keypoints\n![image](https://mediapipe.dev/images/mobile/hand_landmarks.png)## Landmark & body keypoints\nface landmarks: https://github.com/tensorflow/tfjs-models/blob/838611c02f51159afdd77469ce67f0e26b7bbb23/face-landmarks-detection/src/mediapipe-facemesh/keypoints.ts\n\n","metadata":{}},{"cell_type":"code","source":"# Helper functions\n# assign desired colors to landmarks\ndef assign_color(row):\n    if row == 'face':\n        return 'red'\n    elif 'hand' in row:\n        return 'dodgerblue'\n    else:\n        return 'green'\n\n# specifies the plotting order\ndef assign_order(row):\n    if row.type == 'face':\n        return row.landmark_index + 101\n    elif row.type == 'pose':\n        return row.landmark_index + 30\n    elif row.type == 'left_hand':\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index\n    \n    \ndef visualise2d_landmarks(parquet_df):\n    connections = [  \n        [0, 1, 2, 3, 4,],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n\n        \n        [38, 36, 35, 34, 30, 31, 32, 33, 37],\n        [40, 39],\n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        [42, 54, 56, 58, 60, 62, 58],\n        [41, 53, 55, 57, 59, 61, 57],\n        [54, 53],\n\n        \n        [80, 81, 82, 83, 84, ],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100], ]\n\n\n    frames = sorted(set(parquet_df.frame))\n    first_frame = min(frames)\n    parquet_df['color'] = parquet_df.type.apply(lambda row: assign_color(row))\n    parquet_df['plot_order'] = parquet_df.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = parquet_df[parquet_df.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index('plot_order')\n\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = parquet_df[parquet_df.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        traces = [go.Scatter(\n            x=filtered_df['x'],\n            y=filtered_df['y'],\n            mode='markers',\n            marker=dict(\n                color=filtered_df.color,\n                size=9))]\n\n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                    x=filtered_df.loc[seg]['x'],\n                    y=filtered_df.loc[seg]['y'],\n                    mode='lines',\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces = [i for i in range(17)])\n        frames_l.append(frame_data)\n\n    traces = [go.Scatter(\n        x=first_frame_df['x'],\n        y=first_frame_df['y'],\n        mode='markers',\n        marker=dict(\n            color= first_frame_df.color,\n            size=9\n        )\n    )]\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg]['x'],\n            y=first_frame_df.loc[seg]['y'],\n            mode='lines',\n            line=dict(\n                color='black',\n                width=2\n            )\n        )\n        traces.append(trace)\n    fig = go.Figure(\n        data=traces,\n        frames=frames_l\n    )\n\n\n    fig.update_layout(\n        title=\"ASL Sign Visualization\",\n        width=500,\n        height=800,\n        scene={\n            'aspectmode': 'data',\n        },\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [None, {\"frame\": {\"duration\": 100,\n                                                  \"redraw\": True},\n                                        \"fromcurrent\": True,\n                                        \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\":30},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    camera = dict(\n        up=dict(x=0, y=-1, z=0),\n        eye=dict(x=0, y=0, z=2.5)\n    )\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(xaxis = dict(visible=False),\n            yaxis = dict(visible=False),\n    )\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()\n    \n    ","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-21T04:53:19.445005Z","iopub.execute_input":"2023-03-21T04:53:19.445787Z","iopub.status.idle":"2023-03-21T04:53:19.471324Z","shell.execute_reply.started":"2023-03-21T04:53:19.445734Z","shell.execute_reply":"2023-03-21T04:53:19.469834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper functions\n# assign desired colors to landmarks\ndef assign_color(row):\n    if row == 'face':\n        return 'red'\n    elif 'hand' in row:\n        return 'dodgerblue'\n    else:\n        return 'green'\n\n# specifies the plotting order\ndef assign_order(row):\n    if row.type == 'face':\n        return row.landmark_index + 101\n    elif row.type == 'pose':\n        return row.landmark_index + 30\n    elif row.type == 'left_hand':\n        return row.landmark_index + 80\n    else:\n        return row.landmark_index\n    \n    \ndef visualise2d_roi_landmarks(parquet_df):\n    connections = [  # right hand\n        [0, 1, 2, 3, 4, ],\n        [0, 5, 6, 7, 8],\n        [0, 9, 10, 11, 12],\n        [0, 13, 14, 15, 16],\n        [0, 17, 18, 19, 20],\n        \n        [52, 46, 50, 48, 46, 44, 42, 41, 43, 45, 47, 49, 45, 51],\n        # left hand\n        [80, 81, 82, 83, 84, ],\n        [80, 85, 86, 87, 88],\n        [80, 89, 90, 91, 92],\n        [80, 93, 94, 95, 96],\n        [80, 97, 98, 99, 100], ]\n        \n    frames = sorted(set(parquet_df.frame))\n    first_frame = min(frames)\n    parquet_df['color'] = parquet_df.type.apply(lambda row: assign_color(row))\n    parquet_df['plot_order'] = parquet_df.apply(lambda row: assign_order(row), axis=1)\n    first_frame_df = parquet_df[parquet_df.frame == first_frame].copy()\n    first_frame_df = first_frame_df.sort_values([\"plot_order\"]).set_index('plot_order')\n\n\n    frames_l = []\n    for frame in frames:\n        filtered_df = parquet_df[parquet_df.frame == frame].copy()\n        filtered_df = filtered_df.sort_values([\"plot_order\"]).set_index(\"plot_order\")\n        traces = [go.Scatter(\n            x=filtered_df['x'],\n            y=filtered_df['y'],\n            mode='markers',\n            marker=dict(\n                color= filtered_df.color,\n                size=9))]\n\n        for i, seg in enumerate(connections):\n            trace = go.Scatter(\n                    x=filtered_df.loc[seg]['x'],\n                    y=filtered_df.loc[seg]['y'],\n                    mode='lines',\n            )\n            traces.append(trace)\n        frame_data = go.Frame(data=traces, traces = [i for i in range(17)])\n        frames_l.append(frame_data)\n\n    traces = [go.Scatter(\n        x=first_frame_df['x'],\n        y=first_frame_df['y'],\n        mode='markers',\n        marker=dict(\n            color= first_frame_df.color,\n            size=9\n        )\n    )]\n    for i, seg in enumerate(connections):\n        trace = go.Scatter(\n            x=first_frame_df.loc[seg]['x'],\n            y=first_frame_df.loc[seg]['y'],\n            mode='lines',\n            line=dict(\n                color='black',\n                width=2\n            )\n        )\n        traces.append(trace)\n    fig = go.Figure(\n        data=traces,\n        frames=frames_l\n    )\n\n\n    fig.update_layout(\n        title=\"ASL AOI Sign Visualization\",\n        width=500,\n        height=800,\n        scene={\n            'aspectmode': 'data',\n        },\n        updatemenus=[\n            {\n                \"buttons\": [\n                    {\n                        \"args\": [None, {\"frame\": {\"duration\": 100,\n                                                  \"redraw\": True},\n                                        \"fromcurrent\": True,\n                                        \"transition\": {\"duration\": 0}}],\n                        \"label\": \"&#9654;\",\n                        \"method\": \"animate\",\n                    },\n\n                ],\n                \"direction\": \"left\",\n                \"pad\": {\"r\": 100, \"t\": 100},\n                \"font\": {\"size\":30},\n                \"type\": \"buttons\",\n                \"x\": 0.1,\n                \"y\": 0,\n            }\n        ],\n    )\n    camera = dict(\n        up=dict(x=0, y=-1, z=0),\n        eye=dict(x=0, y=0, z=2.5)\n    )\n    fig.update_layout(scene_camera=camera, showlegend=False)\n    fig.update_layout(xaxis = dict(visible=False),\n            yaxis = dict(visible=False),\n    )\n    fig.update_yaxes(autorange=\"reversed\")\n\n    fig.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.473354Z","iopub.execute_input":"2023-03-21T04:53:19.473826Z","iopub.status.idle":"2023-03-21T04:53:19.500308Z","shell.execute_reply.started":"2023-03-21T04:53:19.473787Z","shell.execute_reply":"2023-03-21T04:53:19.499185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_fn = train.query(\"sign== 'wait'\")[\"path\"].values[50]\nexample_landmark = pd.read_parquet(f\"{INPUT_DIR}/{example_fn}\")\n\n#example_landmark\nexample_frame = example_landmark.query(\"frame == 5\")\ngroup_by_df_median = example_frame.dropna().groupby(['type']).median()\n\nexample_landmark_clean = example_landmark.dropna()\nexample_landmark_clean.query(\"type=='right_hand'\").head()\n# What are the unique frames we have\nunique_frames = example_landmark['frame'].unique()\n#print(f\"We have the following unique frames in this landmark example: {unique_frames}, length {len(unique_frames)}\")\n\n# removing pose and face\nexample_landmark_no_pose_face = example_landmark.copy() \nexample_landmark_no_pose_face = example_landmark_no_pose_face[~example_landmark_no_pose_face['type'].isin(['pose']) &\n                                                              ~example_landmark_no_pose_face['type'].isin(['face'])]\n\n# landmark nose, and mouth edges\nexample_landmark_roi_face = example_landmark.copy() \nexample_landmark_roi_face = example_landmark_roi_face[(example_landmark_roi_face['type'].isin(['face'])) & \n                                              (example_landmark_roi_face['landmark_index'].isin(\n                                                  [1, 0, \n                                                   17]))]\n\n\n\nexample_landmark_arms = example_landmark.copy() \nexample_landmark_arms = example_landmark_arms[(example_landmark_arms['type'].isin(['pose'])) & \n                                              (example_landmark_arms['landmark_index'].isin([11,12,13,14,15,16,17,18,19,20,21,22]))]\n\n# get ROI\nexample_landmark_roi = pd.concat([example_landmark_no_pose_face,example_landmark_arms, example_landmark_roi_face], ignore_index=True, sort=False)\n#example_landmark_roi","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.501948Z","iopub.execute_input":"2023-03-21T04:53:19.502794Z","iopub.status.idle":"2023-03-21T04:53:19.571695Z","shell.execute_reply.started":"2023-03-21T04:53:19.502753Z","shell.execute_reply":"2023-03-21T04:53:19.570234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualise2d_roi_landmarks(example_landmark_roi)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T04:53:19.573153Z","iopub.execute_input":"2023-03-21T04:53:19.573503Z","iopub.status.idle":"2023-03-21T04:53:20.099723Z","shell.execute_reply.started":"2023-03-21T04:53:19.573468Z","shell.execute_reply":"2023-03-21T04:53:20.098433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Pre-Processing\n\n### Hypothisis & Assumptions:\n- We are including only the arms from the pose\n- the position of the hands relevant to the face and mouth expression are key features to unlock the sign.\n  One idea is to calculate the distance between each tip of finger and the nose.\n- We only need the outer edges of the mouth.\n\n### List of TODOs: \n- Drop the z coordinates (inaaccurate )\n- Drop all NAN values from the frames. \n- Keep frames with only mouth, tip of the nose, right hand, left hand, and arms. \n- Drop frames where there are no hands.\n- Create a function to process a test sample and then the whole database","metadata":{}},{"cell_type":"markdown","source":"# replicate the following into a function that takes parqute --> return it after doing the follwoing steps:\n1- Do the steps we followed in the visualization step to produce the `example_landmark_roi`\n\n2- make the coordinates origin the tip of the nose.\n\n3- combine it into a matrix","metadata":{}},{"cell_type":"code","source":"import time\n\nt0 = time.time()\n\ntrain_sample = train.iloc[0:]\ndef data_prcess(train_df):\n    #print(train_df_p)\n    # process every parquet file\n    # keep: frames with hand(s), lips, tip of the nose, hands landmarks and only x and y coordinates\n    train_df_p = train_df.copy()\n           \n    features_names = ['r_index_x_m', 'r_index_y_m',\n                     'r_thumb_x_m', 'r_thumb_y_m', 'r_mf_x_m', 'r_mf_y_m', 'r_rf_x_m', 'r_rf_y_m', \n                     'r_pinky_x_m', 'r_pinky_y_m', 'l_index_x_m', 'l_index_y_m',\n                     'l_thumb_x_m', 'l_thumb_y_m', 'l_mf_x_m', 'l_mf_y_m', 'l_rf_x_m', 'l_rf_y_m', \n                     'l_pinky_x_m', 'l_pinky_y_m', 'lips_dist']\n    \n    features_vals = [] # list of lists for every row\n    \n    for index, row in train_df.iterrows():\n        parquet_path = row[\"path\"]\n        sign_df = pd.read_parquet(f\"{INPUT_DIR}/{parquet_path}\")\n        sign_df_process = sign_df.copy()\n        #sign_df_process = sign_df_process[~df['my_col'].isnull()]\n        sign_df_process = sign_df_process.dropna()\n        \n\n         #sign_df_process = sign_df_process.dropna().drop('z', axis=1)\n        \n        # right hand features\n        r_index_x_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([8]))]['x'].median()\n        r_index_y_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([8]))]['y'].median()\n        r_thumb_x_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([4]))]['x'].median()\n        r_thumb_y_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([4]))]['y'].median()  \n        r_mf_x_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                   (sign_df_process['landmark_index'].isin([12]))]['x'].median()  \n        r_mf_y_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                   (sign_df_process['landmark_index'].isin([12]))]['y'].median()  \n        r_rf_x_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                   (sign_df_process['landmark_index'].isin([16]))]['x'].median()  \n        r_rf_y_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                   (sign_df_process['landmark_index'].isin([16]))]['y'].median()  \n        r_pinky_x_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([20]))]['x'].median()  \n        r_pinky_y_m = sign_df_process[(sign_df_process['type'].isin(['right_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([20]))]['y'].median()  \n        \n        # left hand features\n        l_index_x_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([8]))]['x'].median() \n        l_index_y_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([8]))]['y'].median() \n        l_thumb_x_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([4]))]['x'].median() \n        l_thumb_y_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([4]))]['y'].median() \n        l_mf_x_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([12]))]['x'].median() \n        l_mf_y_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([12]))]['y'].median() \n        l_rf_x_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([16]))]['x'].median() \n        l_rf_y_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([16]))]['y'].median() \n        l_pinky_x_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([20]))]['x'].median() \n        l_pinky_y_m = sign_df_process[(sign_df_process['type'].isin(['left_hand'])) & \n                                      (sign_df_process['landmark_index'].isin([20]))]['y'].median() \n        \n        up_center_lip_x = sign_df_process[(sign_df_process['type'].isin(['face'])) & \n                                      (sign_df_process['landmark_index'].isin([0]))]['x'].median()\n        up_center_lip_y = sign_df_process[(sign_df_process['type'].isin(['face'])) & \n                                      (sign_df_process['landmark_index'].isin([0]))]['y'].median()\n        down_center_lip_x = sign_df_process[(sign_df_process['type'].isin(['face'])) & \n                                      (sign_df_process['landmark_index'].isin([17]))]['x'].median()\n        \n        down_center_lip_y = sign_df_process[(sign_df_process['type'].isin(['face'])) & \n                                      (sign_df_process['landmark_index'].isin([17]))]['y'].median()\n        \n        a = np.array((up_center_lip_x, up_center_lip_y))\n        b = np.array((down_center_lip_x, down_center_lip_y))\n        lips_dist = np.linalg.norm(a-b)\n        #mouth_area_median\n                    \n        features_vals.append([r_index_x_m, r_index_y_m,\n                             r_thumb_x_m, r_thumb_y_m, r_mf_x_m, r_mf_y_m, r_rf_x_m, r_rf_y_m, \n                             r_pinky_x_m, r_pinky_y_m, l_index_x_m, l_index_y_m,\n                             l_thumb_x_m, l_thumb_y_m, l_mf_x_m, l_mf_y_m, l_rf_x_m, l_rf_y_m, \n                             l_pinky_x_m, l_pinky_y_m, lips_dist])\n                            \n        df1 = pd.DataFrame(features_vals, columns=features_names)\n        df2 = pd.concat([train_df_p, df1], axis=1)\n        df2 = df2.fillna(0.0)\n       \n    # Create the meta data calculate the average of x, y points\n        \n    return (df2)\n\n\ntrain_meta_data = data_prcess(train_sample)\ntrain_meta_data.to_parquet(\"/kaggle/working/train_meta_data.parquet\")\nt1 = time.time()\n\ntotal = t1-t0","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:01:59.775350Z","iopub.execute_input":"2023-03-21T09:01:59.775958Z","iopub.status.idle":"2023-03-21T09:02:00.484778Z","shell.execute_reply.started":"2023-03-21T09:01:59.775899Z","shell.execute_reply":"2023-03-21T09:02:00.483432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:02:01.425953Z","iopub.execute_input":"2023-03-21T09:02:01.427294Z","iopub.status.idle":"2023-03-21T09:02:01.434728Z","shell.execute_reply.started":"2023-03-21T09:02:01.427241Z","shell.execute_reply":"2023-03-21T09:02:01.433207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta_data","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:00:31.594914Z","iopub.execute_input":"2023-03-21T09:00:31.596446Z","iopub.status.idle":"2023-03-21T09:00:31.632482Z","shell.execute_reply.started":"2023-03-21T09:00:31.596394Z","shell.execute_reply":"2023-03-21T09:00:31.630151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation\n- from evaluation page","metadata":{}},{"cell_type":"code","source":"test = pd.read_parquet(f\"{INPUT_DIR}/train_landmark_files/32319/1000278229.parquet\")\nlen(test.query(\"frame == 44\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load_relevant_data_subset(f\"{INPUT_DIR}/train_landmark_files/32319/1000278229.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''import tflite_runtime.interpreter as tflite\ninterpreter = tflite.Interpreter(model_path)\n\nfound_signatures = list(interpreter.get_signature_list().keys())\n\nif REQUIRED_SIGNATURE not in found_signatures:\n    raise KernelEvalException('Required input signature not found.')\n\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\noutput = prediction_fn(inputs=frames)\nsign = np.argmax(output[\"outputs\"])'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}