{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Google - Isolated Sign Language Recognition","metadata":{}},{"cell_type":"code","source":"!pip install nb_black --quiet\n!pip install mediapipe --quiet\n%load_ext lab_black","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:36:53.458220Z","iopub.execute_input":"2023-03-01T15:36:53.458663Z","iopub.status.idle":"2023-03-01T15:37:29.294908Z","shell.execute_reply.started":"2023-03-01T15:36:53.458625Z","shell.execute_reply":"2023-03-01T15:37:29.293525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport mediapipe as mp","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:18.197446Z","iopub.execute_input":"2023-03-01T15:38:18.197930Z","iopub.status.idle":"2023-03-01T15:38:18.724457Z","shell.execute_reply.started":"2023-03-01T15:38:18.197887Z","shell.execute_reply":"2023-03-01T15:38:18.723148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config ","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/asl-signs\"\ntrain_csv = pd.read_csv(f\"{BASE_DIR}/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:19.169365Z","iopub.execute_input":"2023-03-01T15:38:19.169773Z","iopub.status.idle":"2023-03-01T15:38:19.392581Z","shell.execute_reply.started":"2023-03-01T15:38:19.169737Z","shell.execute_reply":"2023-03-01T15:38:19.391365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA ","metadata":{}},{"cell_type":"markdown","source":"## Exploring train_csv file","metadata":{}},{"cell_type":"code","source":"train_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:21.155376Z","iopub.execute_input":"2023-03-01T15:38:21.155775Z","iopub.status.idle":"2023-03-01T15:38:21.186354Z","shell.execute_reply.started":"2023-03-01T15:38:21.155741Z","shell.execute_reply":"2023-03-01T15:38:21.185392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Participants \n\n* There are 21 different participant representing between 3300 and 5000 signs. \n* There are 2 participants that do not represent all the signs (Participant_id 30680, Participant_id 25571)\n* It seem that the signs and frequence of representation is arbitrary, like they did random signs that they knew (there is a participant who didn't do all the signs and did more than others that did them):\n    - Maximun value: 20 - 32\n    - Minimun value: 0 - 15\n    - Medium value: 14 -19","metadata":{}},{"cell_type":"code","source":"train_csv.participant_id.value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:22.742487Z","iopub.execute_input":"2023-03-01T15:38:22.743121Z","iopub.status.idle":"2023-03-01T15:38:23.126582Z","shell.execute_reply.started":"2023-03-01T15:38:22.743085Z","shell.execute_reply":"2023-03-01T15:38:23.125385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def brief(participant_id):\n    participant_id_query = train_csv.query(\n        f\"participant_id == {participant_id}\"\n    ).sign.value_counts()\n    maximun = participant_id_query.apply(lambda g: g == participant_id_query.max())\n    minimun = participant_id_query.apply(lambda g: g == participant_id_query.min())\n\n    print(\n        f\"Participant_id {participant_id}\\n\"\n        f\"Amount of different signs represented: {len(participant_id_query)}\\n\"\n        f\"Total amount of sign represented: {participant_id_query.sum()}\\n\"\n        f\"Most represented sign: {maximun[maximun].index.values} with {participant_id_query.max()} representations\\n\"\n        f\"Least represented sign: {minimun[minimun].index.values} with {participant_id_query.min()} representations\\n\"\n        f\"Avg number of representations: {participant_id_query.mean():.2f}\\n\"\n        f\"Median value of representations: {participant_id_query.median()}\\n\\n\"\n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:23.153086Z","iopub.execute_input":"2023-03-01T15:38:23.153884Z","iopub.status.idle":"2023-03-01T15:38:23.166036Z","shell.execute_reply.started":"2023-03-01T15:38:23.153837Z","shell.execute_reply":"2023-03-01T15:38:23.164734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train_csv.participant_id.value_counts().index:\n    brief(i)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:24.229006Z","iopub.execute_input":"2023-03-01T15:38:24.229828Z","iopub.status.idle":"2023-03-01T15:38:24.841556Z","shell.execute_reply.started":"2023-03-01T15:38:24.229772Z","shell.execute_reply":"2023-03-01T15:38:24.840199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Signs  \n\n* There are a total of 250 signs\n* The most represented words are **listen**, **look** and **donkey** with more than 400 representations \n* The least represented words are **zipper**, **vacuum** and **beside** with around 300 representations ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2)\ntrain_csv.sign.value_counts().head(10).plot(kind=\"bar\", ax=ax[0])\ntrain_csv.sign.value_counts().tail(10).plot(kind=\"bar\", ax=ax[1])\nax[0].title.set_text(\"Top 10 signs\")\nax[1].title.set_text(\"Bottom 10 signs\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:24.979577Z","iopub.execute_input":"2023-03-01T15:38:24.980352Z","iopub.status.idle":"2023-03-01T15:38:25.349122Z","shell.execute_reply.started":"2023-03-01T15:38:24.980296Z","shell.execute_reply":"2023-03-01T15:38:25.347846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring one sign (Parquet landmark data)","metadata":{}},{"cell_type":"code","source":"example_landmark_data = pd.read_parquet(f\"{BASE_DIR}/{train_csv.path[0]}\")\nexample_landmark_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:59:36.512701Z","iopub.execute_input":"2023-03-01T15:59:36.513394Z","iopub.status.idle":"2023-03-01T15:59:36.550980Z","shell.execute_reply.started":"2023-03-01T15:59:36.513305Z","shell.execute_reply":"2023-03-01T15:59:36.549800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's evaluate the amount of frames per parquet data\n\n* Frames are compromised between 0 and 546:\n    - Maximun frame window: 536\n    - Minimun frame window: 1\n    - Average frame window: 36,94\n* They can start and end at any frame window \n    - Start range: 0 - 484\n    - End range:  1 - 546\n* Since the first frame they detect until the last one they will have data for all the intermediate frames (even though there is no data available = NaN)\n","metadata":{}},{"cell_type":"code","source":"frames = []\nframe_duration = []\nNUMBER_SAMPLES = len(train_csv)\nlandmarks = []\n\nfor i, d in tqdm(train_csv.iterrows()):\n    file_path = d[\"path\"]\n    landmark_data = pd.read_parquet(f\"{BASE_DIR}/{file_path}\")\n    frames.append(\n        [\n            landmark_data[\"frame\"].unique()[0],\n            landmark_data[\"frame\"].unique()[-1],\n        ]\n    )\n    frame_duration.append(landmark_data[\"frame\"].unique())\n\n    if i > NUMBER_SAMPLES:\n        break\n        \nframes = np.array(frames).reshape(-1, 2)\nprint(\n    f\"Frame range: {min(frames[:,0])} - {max(frames[:,1])}\\n\"\n    f\"Start frame range: {min(frames[:,0])} - {max(frames[:,0])} --> mean: {np.mean(frames[:,0])}\\n\"\n    f\"End frame range: {min(frames[:,1])} - {max(frames[:,1])} --> mean: {np.mean(frames[:,1])}\\n\"\n)\n\ntrue_frames = np.array(frames[:, 1]) - np.array(frames[:, 0])\nprint(\n    f\"Maximun amount of frames in a single parquet file: {max(true_frames)}\\n\"\n    f\"Minimun amount of frames in a single parquet file: {min(true_frames)}\\n\"\n    f\"Average amount of frames in a single parquet file: {np.mean(true_frames):.2f}\\n\"\n)\nprint(\n    f\"Frames of the longest frame file in a single parquet file:\\n{frame_duration[np.argmax(true_frames)]}\\n\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:40.709118Z","iopub.execute_input":"2023-03-01T15:38:40.709537Z","iopub.status.idle":"2023-03-01T15:38:43.108121Z","shell.execute_reply.started":"2023-03-01T15:38:40.709499Z","shell.execute_reply":"2023-03-01T15:38:43.106600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### What about the landmarks \n\n* There are 4 different types of landmarks (Face, left hand, righ hand and pose)\n    - Face 468 points\n    - Left hand: 21 points \n    - Right hand: 21 points \n    - Pose: 33 points \n* All the frames content a value for each keypoint of the 4 different landmarks (if there is no information they will output Nan value for x,y,z):\n    - 55.09% of the parquet data has NaN values for all the left hand keypoints\n    - 40.68% of the parquet data has NaN values for all the right hand keypoints\n    - 4.23% of parquet data has values for both left and right keypoints\n    \n**As we already suspected they videos seem to have been recorded holding the phone with one hand and doing the signs with the other one**\n\n   - This means that 12 out of 21 participants are right hand\n   - That 4% where there is data about both hands could be a bug from the MediaPipe holistic model.\n","metadata":{}},{"cell_type":"code","source":"print(\"Landmarks:\")\nfor i in example_landmark_data.type.unique():\n    a = example_landmark_data.query(f'type == \"{i}\"').landmark_index.unique()\n    print(f\"\\t{i} has {len(a)} different keypoints\")","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:38:43.316270Z","iopub.execute_input":"2023-03-01T15:38:43.316683Z","iopub.status.idle":"2023-03-01T15:38:43.340232Z","shell.execute_reply.started":"2023-03-01T15:38:43.316649Z","shell.execute_reply":"2023-03-01T15:38:43.338747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUMBER_SAMPLES = len(train_csv.participant_id.unique())\nright_hand, left_hand, both_hands = 0, 0, 0\nright, left = True, True\n\n\nhands_df = pd.DataFrame(\n    {\"participant\": [], \"sign\": [], \"right_hand\": [], \"left_hand\": [], \"both_hands\": []}\n)\nfor i, participant_id in enumerate(tqdm(train_csv.participant_id.unique())):\n    participant = train_csv.query(f\"participant_id == {participant_id}\")\n    for sign in participant.sign.unique():\n        for _, d in participant.query(f'sign == \"{sign}\"').iterrows():\n            file_path = d[\"path\"]\n            landmark_data = pd.read_parquet(f\"{BASE_DIR}/{file_path}\")\n            for hand in [\"left_hand\", \"right_hand\"]:\n                values = landmark_data.query(f'type == \"{hand}\"')\n                if values[\"x\"].isnull().sum() == len(values):\n                    if hand == \"left_hand\":\n                        left = False\n                    elif hand == \"right_hand\":\n                        right = False\n\n            if right and left:\n                both_hands += 1\n            else:\n                if right:\n                    right_hand += 1\n                elif left:\n                    left_hand += 1\n            right, left = True, True\n\n        hands_df = hands_df.append(\n            {\n                \"participant\": participant_id,\n                \"sign\": sign,\n                \"right_hand\": right_hand,\n                \"left_hand\": left_hand,\n                \"both_hands\": both_hands,\n            },\n            ignore_index=True,\n        )\n        right_hand, left_hand, both_hands = 0, 0, 0\n    if i > NUMBER_SAMPLES:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:40:57.435060Z","iopub.execute_input":"2023-03-01T15:40:57.436585Z","iopub.status.idle":"2023-03-01T15:46:08.965969Z","shell.execute_reply.started":"2023-03-01T15:40:57.436522Z","shell.execute_reply":"2023-03-01T15:46:08.964896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hands_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:55:19.781777Z","iopub.execute_input":"2023-03-01T15:55:19.783551Z","iopub.status.idle":"2023-03-01T15:55:19.812196Z","shell.execute_reply.started":"2023-03-01T15:55:19.783470Z","shell.execute_reply":"2023-03-01T15:55:19.810659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signs_df = hands_df.groupby([\"sign\"]).sum().drop(\"participant\", axis=1)\nsigns_df[\"total\"] = signs_df.sum(axis=1)\nsigns_df = signs_df.sort_values(by=[\"total\"], ascending=False)\nsigns_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:55:20.135243Z","iopub.execute_input":"2023-03-01T15:55:20.136602Z","iopub.status.idle":"2023-03-01T15:55:20.166391Z","shell.execute_reply.started":"2023-03-01T15:55:20.136541Z","shell.execute_reply":"2023-03-01T15:55:20.165410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\n    f\"{signs_df.sum()[0]/signs_df.sum()[3]*100:.2f}% of signs represented with only right hand\\n\"\n    f\"{signs_df.sum()[1]/signs_df.sum()[3]*100:.2f}% of signs represented with only left hand\\n\"\n    f\"{signs_df.sum()[2]/signs_df.sum()[3]*100:.2f}% o signs represented with both hands\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:55:21.742916Z","iopub.execute_input":"2023-03-01T15:55:21.743818Z","iopub.status.idle":"2023-03-01T15:55:21.758208Z","shell.execute_reply.started":"2023-03-01T15:55:21.743771Z","shell.execute_reply":"2023-03-01T15:55:21.756700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2)\nsigns_df.drop(\"total\", axis=1).head(5).plot(kind=\"bar\", ax=ax[0])\nsigns_df.drop(\"total\", axis=1).tail(5).plot(kind=\"bar\", ax=ax[1])\nax[0].title.set_text(\"Top 10 signs\")\nax[1].title.set_text(\"Bottom 10 signs\")\n\nax[0].legend(loc=\"center left\", bbox_to_anchor=(1, 1))\nax[1].legend(loc=\"center left\", bbox_to_anchor=(1, 1))\n\nplt.subplots_adjust(right=1.5, wspace=0.7)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T15:55:24.058859Z","iopub.execute_input":"2023-03-01T15:55:24.059808Z","iopub.status.idle":"2023-03-01T15:55:24.488183Z","shell.execute_reply.started":"2023-03-01T15:55:24.059761Z","shell.execute_reply":"2023-03-01T15:55:24.486899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plotting an example ","metadata":{}},{"cell_type":"code","source":"def plotting_frame(csv_file, parquet_file_path, frame_index):\n    mp_hands = mp.solutions.hands\n    mp_pose = mp.solutions.pose\n    mp_face_mesh = mp.solutions.face_mesh\n\n    participant_id = csv_file.query(f'path == \"{parquet_file_path}\"').participant_id[0]\n    parquet_file_df = pd.read_parquet(f\"{BASE_DIR}/{parquet_file_path}\")\n\n    # y axis is inverted so we need to multiply it *-1\n    parquet_file_df[\"y_\"] = parquet_file_df[\"y\"] * -1\n    frame_sample = parquet_file_df.query(f\"frame == {frame_index}\")\n    fig, ax = plt.subplots(figsize=(5, 5))\n\n    # Getting hands plot\n    for hand in [\"left_hand\", \"right_hand\"]:\n        hand_sample = frame_sample.query(f\"type == '{hand}'\")\n        #     ax.scatter(example_hand[\"x\"], example_hand[\"y_\"]) --> uncomment if you want to get also the data points\n\n        for connection in mp_hands.HAND_CONNECTIONS:\n            point_a = connection[0]\n            point_b = connection[1]\n            x1, y1 = hand_sample[hand_sample[\"landmark_index\"] == point_a][\n                [\"x\", \"y_\"]\n            ].values[0]\n            x2, y2 = hand_sample[hand_sample[\"landmark_index\"] == point_b][\n                [\"x\", \"y_\"]\n            ].values[0]\n            plt.plot(\n                [x1, x2], [y1, y2], color=\"purple\" if hand == \"left_hand\" else \"orange\"\n            )\n\n    pose_sample = frame_sample.query('type == \"pose\"')\n    # ax.scatter(example_pose[\"x\"], example_pose[\"y_\"])\n\n    for connection in mp_pose.POSE_CONNECTIONS:\n        point_a = connection[0]\n        point_b = connection[1]\n        x1, y1 = pose_sample[pose_sample[\"landmark_index\"] == point_a][\n            [\"x\", \"y_\"]\n        ].values[0]\n        x2, y2 = pose_sample[pose_sample[\"landmark_index\"] == point_b][\n            [\"x\", \"y_\"]\n        ].values[0]\n        plt.plot([x1, x2], [y1, y2], color=\"green\")\n\n    face_sample = frame_sample.query('type == \"face\"')\n    # ax.scatter(example_pose[\"x\"], example_pose[\"y_\"])\n\n    for connection in mp_face_mesh.FACEMESH_CONTOURS:\n        point_a = connection[0]\n        point_b = connection[1]\n        x1, y1 = face_sample[face_sample[\"landmark_index\"] == point_a][\n            [\"x\", \"y_\"]\n        ].values[0]\n        x2, y2 = face_sample[face_sample[\"landmark_index\"] == point_b][\n            [\"x\", \"y_\"]\n        ].values[0]\n        plt.plot([x1, x2], [y1, y2], color=\"red\")\n\n    plt.title(f\"Participant id: {participant_id}, Frame: {frame_index}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-01T16:20:52.016296Z","iopub.execute_input":"2023-03-01T16:20:52.016787Z","iopub.status.idle":"2023-03-01T16:20:52.071272Z","shell.execute_reply.started":"2023-03-01T16:20:52.016749Z","shell.execute_reply":"2023-03-01T16:20:52.069817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plotting_frame(train_csv, train_csv.path[0], 20)","metadata":{"execution":{"iopub.status.busy":"2023-03-01T16:20:52.268464Z","iopub.execute_input":"2023-03-01T16:20:52.268912Z","iopub.status.idle":"2023-03-01T16:20:53.134866Z","shell.execute_reply.started":"2023-03-01T16:20:52.268872Z","shell.execute_reply":"2023-03-01T16:20:53.133707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hands_df.to_csv('hands_df.csv',index=False)","metadata":{},"execution_count":null,"outputs":[]}]}