{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Sign Language Competition\nThe goal of this competition is to classify Americal Sign Language(ASL) signs.\nThe landmarks were extracted from raw videos with mediapipe holistic model.\nWe are supposed to predict the sign from this data","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"!pip install mediapipe --quiet\n!pip install itables --quiet","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:06:50.091364Z","iopub.execute_input":"2023-04-13T13:06:50.092117Z","iopub.status.idle":"2023-04-13T13:07:16.674057Z","shell.execute_reply.started":"2023-04-13T13:06:50.092073Z","shell.execute_reply":"2023-04-13T13:07:16.672733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom itables import init_notebook_mode\ninit_notebook_mode(all_interactive=True,connected=True)\n\nimport os\nimport json\nimport itertools\nimport tensorflow as tf\n\nimport torch\nimport torch.nn.functional as F\nimport torch.nn as nn\n\nimport mediapipe as mp\nfrom mediapipe.framework.formats import landmark_pb2\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\nfrom matplotlib import animation\nfrom pathlib import Path\nimport IPython\nfrom IPython import display\nfrom IPython.core.display import display, HTML,Javascript\nfrom IPython.display import Markdown as md\n\nplt.style.use(\"seaborn-colorblind\")\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:16.676524Z","iopub.execute_input":"2023-04-13T13:07:16.676905Z","iopub.status.idle":"2023-04-13T13:07:29.725587Z","shell.execute_reply.started":"2023-04-13T13:07:16.676866Z","shell.execute_reply":"2023-04-13T13:07:29.724125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# install nb_black for autoformating\n!pip install nb_black --quiet\n%load_ext lab_black","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:29.727199Z","iopub.execute_input":"2023-04-13T13:07:29.727576Z","iopub.status.idle":"2023-04-13T13:07:43.994947Z","shell.execute_reply.started":"2023-04-13T13:07:29.727541Z","shell.execute_reply":"2023-04-13T13:07:43.993602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation \nThe evaluation metric for this contest is simple classification accuracy.\n\n","metadata":{}},{"cell_type":"markdown","source":"# data EDA","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"../input/asl-signs/\"\ntrain = pd.read_csv(f\"{BASE_DIR}train.csv\")\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:43.998240Z","iopub.execute_input":"2023-04-13T13:07:43.998615Z","iopub.status.idle":"2023-04-13T13:07:44.228225Z","shell.execute_reply.started":"2023-04-13T13:07:43.998575Z","shell.execute_reply":"2023-04-13T13:07:44.227029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    INPUT_ROOT = \"../input/asl-signs/\"\n    OUTPUT_ROOT = \"../kaggle/working\"\n    INDEX_MAP_FILE = f\"{INPUT_ROOT}sign_to_prediction_index_map.json\"\n\n    TRAIN_FILE = f\"{INPUT_ROOT}/train.csv\"\n    INDEX = \"sequence_id\"\n    ROW_ID = \"row_id\"\n\n\nLANDMARK_FILES_DIR = f\"{CFG.INPUT_ROOT}train_landmark_files\"\nlabel_map = json.load(\n    open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\")\n)\ntrain_df = pd.read_csv(\n    \"/kaggle/input/gislr-extended-train-dataframe/extended_train.csv\"\n)\ntrain_df[\"label\"] = train_df[\"sign\"].map(label_map)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:44.229914Z","iopub.execute_input":"2023-04-13T13:07:44.230418Z","iopub.status.idle":"2023-04-13T13:07:45.030701Z","shell.execute_reply.started":"2023-04-13T13:07:44.230342Z","shell.execute_reply":"2023-04-13T13:07:45.029621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helpers\n\nROWS_PER_FRAME = 543\n\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = [\"x\", \"y\", \"z\"]\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\n\n# a FUNCTION TO read json file\ndef read_json(file_path=CFG.INDEX_MAP_FILE):\n    try:\n        with open(file_path, \"r\") as file:\n            json_data = json.load(file)\n            return json_data\n    except FileNotFoundError:\n        raise FileNotFoundError(f\"File not found as {file_path}\")\n    except ValueError:\n        raise ValueError(f\"Invalid JSON Data at {file_path}\")\n\n\n# write a function to load training data and set proper index\ndef load_train(file_path=CFG.TRAIN_FILE):\n    train = pd.read_csv(file_path)\n    train.set_index(CFG.INDEX)\n    return train\n\n\n# write a function which will have input root, and parquet file path as an argument and then load the landmarks\ndef load_landmarks_path(file_path, input_root=CFG.INPUT_ROOT):\n    landmarks = pd.load_parquet(f\"{input_root}{file_path}\")\n    landmarks.set_index(CFG.ROW_ID)\n    return landmarks\n\n\n# write a function which will receive a sequence_id from training data as an input and using path for that particular sequence_id load the parquet file\ndef load_landmarks_id(sequence_id, train_data):\n    landmarks = pd.read_parquet(train_data.loc[sequence_id][\"path\"])\n    return landmarks","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:45.031995Z","iopub.execute_input":"2023-04-13T13:07:45.032310Z","iopub.status.idle":"2023-04-13T13:07:45.053318Z","shell.execute_reply.started":"2023-04-13T13:07:45.032280Z","shell.execute_reply":"2023-04-13T13:07:45.051902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`get_ids` - function returns participant id and sequence id\n\n`draw_data` -  \n1. function takes participant_id, sequence_id(returned by get_ids function) and train_data as and argument\n2. for a particular participant_id landmarks file is loaded using load_landmarks_id function\n3. frame_ids is the list of unique frames for that participant id\n4. Then we iterate through this frame_ids and use draw landmarks function for each type(right_hand,left_hand,pose,face)\n5. `draw_landmarks` - inputs taken\n    a. landmarks - landmarks we loaded earlier in the above function\n    \n    b. image - a blank image\n    \n    c. frame_id - one of the frame_ids we started iterating through earlier\n    \n    d. landmark_type = right_hand,left_hand,pose,face)\n    \n    e. connection_type = for right and left hand - mp_hands.HAND_CONNECTIONS\n                         for face - mp_face_mesh.FACEMESH_CONTOURS\n                         for pose - mp_pose.POSE_CONTOURS\n    f. landmark_color\n    \n    g. connection_color\n    \n    h. thickness\n    \n    i. circle_radius\n    \n","metadata":{}},{"cell_type":"code","source":"train_data = load_train()\n\n\nmp_drawing = mp.solutions.drawing_utils\nmp_face_mesh = mp.solutions.face_mesh\nmp_hands = mp.solutions.hands\nmp_pose = mp.solutions.pose\n\nCONTOURS = list(itertools.chain(*mp_face_mesh.FACEMESH_CONTOURS))","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:45.054674Z","iopub.execute_input":"2023-04-13T13:07:45.055588Z","iopub.status.idle":"2023-04-13T13:07:45.196306Z","shell.execute_reply.started":"2023-04-13T13:07:45.055549Z","shell.execute_reply":"2023-04-13T13:07:45.195019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_ids(df, row_id):\n    participant_id = df.participant_id.values[row_id]\n    sequence_id = df.sequence_id.values[row_id]\n    return participant_id, sequence_id\n\n\ndef blank_image(height, width):\n    return np.zeros((height, width, 3), np.uint8)\n\n\ndef draw_landmarks(\n    data,\n    image,\n    frame_id,\n    landmark_type,\n    connection_type,\n    landmark_color=(255, 0, 0),\n    connection_color=(0, 20, 255),\n    thickness=2,\n    circle_radius=1,\n):\n    \"\"\"Draws landmarks\"\"\"\n    df = data.groupby([\"frame\", \"type\"]).get_group((frame_id, landmark_type)).copy()\n    if landmark_type == \"face\":\n        df.loc[~df[\"landmark_index\"].isin(CONTOURS), \"x\"] = float(\n            \"NaN\"\n        )  # -1*df[~df['landmark_index'].isin(CONTOURS)]['x'].values\n\n    landmarks = [\n        landmark_pb2.NormalizedLandmark(x=lm.x, y=lm.y, z=lm.z)\n        for idx, lm in df.iterrows()\n    ]\n    landmark_list = landmark_pb2.NormalizedLandmarkList(landmark=landmarks)\n    # print(len(landmark_list.landmark))\n    mp_drawing.draw_landmarks(\n        image=image,\n        landmark_list=landmark_list,\n        connections=connection_type,\n        landmark_drawing_spec=mp_drawing.DrawingSpec(\n            color=landmark_color, thickness=thickness, circle_radius=circle_radius\n        ),\n        connection_drawing_spec=mp_drawing.DrawingSpec(\n            color=connection_color, thickness=thickness, circle_radius=circle_radius\n        ),\n    )\n    return image\n\n\ndef draw_data(participant_id, sequence_id, train_data):\n    height, width = 500, 500\n    landmarks = load_landmarks_id(participant_id, train_data)\n    frames = landmarks[\"frame\"].unique().tolist()\n\n    FIG = make_subplots(\n        rows=1,\n        cols=5,\n    )\n    for i, frame in enumerate(frames):\n        r_hand = draw_landmarks(\n            landmarks,\n            image=blank_image(height, width),\n            frame_id=frame,\n            landmark_type=\"right_hand\",\n            connection_type=mp_hands.HAND_CONNECTIONS,\n            landmark_color=(255, 255, 255),\n            connection_color=(0, 255, 0),\n            thickness=1,\n            circle_radius=1,\n        )\n        l_hand = draw_landmarks(\n            landmarks,\n            image=blank_image(height, width),\n            frame_id=frame,\n            landmark_type=\"left_hand\",\n            connection_type=mp_hands.HAND_CONNECTIONS,\n            landmark_color=(255, 255, 255),\n            connection_color=(255, 0, 0),\n            thickness=1,\n            circle_radius=1,\n        )\n        face = draw_landmarks(\n            landmarks,\n            image=blank_image(height, width),\n            frame_id=frame,\n            landmark_type=\"face\",\n            connection_type=mp_face_mesh.FACEMESH_CONTOURS,\n            landmark_color=(255, 255, 255),\n            connection_color=(0, 255, 0),\n            thickness=1,\n            circle_radius=1,\n        )\n        pose = draw_landmarks(\n            landmarks,\n            image=blank_image(height, width),\n            frame_id=frame,\n            landmark_type=\"pose\",\n            connection_type=mp_pose.POSE_CONNECTIONS,\n            landmark_color=(255, 255, 255),\n            connection_color=(0, 0, 255),\n            thickness=1,\n            circle_radius=1,\n        )\n\n        FIG.add_trace(px.imshow(face).data[0], row=1, col=1)\n        FIG.add_trace(px.imshow(pose).data[0], row=1, col=2)\n        FIG.add_trace(px.imshow(r_hand).data[0], row=1, col=3)\n        FIG.add_trace(px.imshow(l_hand).data[0], row=1, col=4)\n        FIG.add_trace(\n            px.imshow(face + pose + l_hand + r_hand, aspect=\"auto\").data[0],\n            row=1,\n            col=5,\n        )\n\n    return FIG","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:45.198444Z","iopub.execute_input":"2023-04-13T13:07:45.198899Z","iopub.status.idle":"2023-04-13T13:07:45.556534Z","shell.execute_reply.started":"2023-04-13T13:07:45.198853Z","shell.execute_reply":"2023-04-13T13:07:45.555439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"participant_id, sequence_id = get_ids(train_df.query('sign==\"sleepy\"'), 10)\nfig = draw_data(participant_id, sequence_id, train_df)\nfig","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:09:38.416363Z","iopub.execute_input":"2023-04-13T13:09:38.416816Z","iopub.status.idle":"2023-04-13T13:09:45.775508Z","shell.execute_reply.started":"2023-04-13T13:09:38.416777Z","shell.execute_reply":"2023-04-13T13:09:45.774308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train.csv has path to each parquet file, participant id, sequence id, and sign\n# What signs are we trying to predict?","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 8))\ntrain[\"sign\"].value_counts().head(50).sort_values(ascending=True).plot(\n    kind=\"barh\", ax=ax, title=\"Top 50 Signs\"\n)\nax.set_xlabel(\"Number of Training Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:51.685289Z","iopub.execute_input":"2023-04-13T13:07:51.685642Z","iopub.status.idle":"2023-04-13T13:07:52.423907Z","shell.execute_reply.started":"2023-04-13T13:07:51.685610Z","shell.execute_reply":"2023-04-13T13:07:52.422936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 8))\ntrain[\"sign\"].value_counts().tail(50).sort_values(ascending=True).plot(\n    kind=\"barh\", ax=ax, title=\"Bottom 50 Signs\"\n)\nax.set_xlabel(\"Number of Training Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:52.425368Z","iopub.execute_input":"2023-04-13T13:07:52.426098Z","iopub.status.idle":"2023-04-13T13:07:53.162526Z","shell.execute_reply.started":"2023-04-13T13:07:52.426059Z","shell.execute_reply":"2023-04-13T13:07:53.161214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parquet Landmark Data\n- Each parquet file in the path\n    - path = train_landmark_files/[participant_id]/[sequence_id].parquet\n    - parquet associated sign can be found in train.csv","metadata":{"execution":{"iopub.status.busy":"2023-04-10T18:28:51.751987Z","iopub.execute_input":"2023-04-10T18:28:51.752339Z","iopub.status.idle":"2023-04-10T18:28:51.757893Z","shell.execute_reply.started":"2023-04-10T18:28:51.752309Z","shell.execute_reply":"2023-04-10T18:28:51.756452Z"}}},{"cell_type":"code","source":"train.query('sign == \"listen\"')","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:53.164360Z","iopub.execute_input":"2023-04-13T13:07:53.164763Z","iopub.status.idle":"2023-04-13T13:07:53.216142Z","shell.execute_reply.started":"2023-04-13T13:07:53.164729Z","shell.execute_reply":"2023-04-13T13:07:53.214654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_listen = train.query('sign==\"listen\"')[\"path\"].values[0]\nlisten_landmark = pd.read_parquet(f\"{BASE_DIR}{first_listen}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:53.218518Z","iopub.execute_input":"2023-04-13T13:07:53.219034Z","iopub.status.idle":"2023-04-13T13:07:53.256156Z","shell.execute_reply.started":"2023-04-13T13:07:53.218980Z","shell.execute_reply":"2023-04-13T13:07:53.254786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"listen_landmark","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:53.257997Z","iopub.execute_input":"2023-04-13T13:07:53.258479Z","iopub.status.idle":"2023-04-13T13:07:53.377054Z","shell.execute_reply.started":"2023-04-13T13:07:53.258429Z","shell.execute_reply":"2023-04-13T13:07:53.375639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_frames = listen_landmark[\"frame\"].nunique()\nunique_types = listen_landmark[\"type\"].nunique()\n\nprint(f\"This file has total of {unique_frames} unique frames\")\nprint(\n    f\"This file has total of {unique_types} unique types which contain {listen_landmark['type'].unique()}\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:53.378880Z","iopub.execute_input":"2023-04-13T13:07:53.380149Z","iopub.status.idle":"2023-04-13T13:07:53.396292Z","shell.execute_reply.started":"2023-04-13T13:07:53.380080Z","shell.execute_reply":"2023-04-13T13:07:53.394237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compare number of parquet files\n- We noticed that number of frames in each parquet file is different\n- for the examples taken all 4 types were present","metadata":{}},{"cell_type":"code","source":"listen_files = train.query('sign==\"listen\"')[\"path\"].values\nfor i, file in enumerate(listen_files):\n    listen_landmark = pd.read_parquet(f\"{BASE_DIR}{file}\")\n    unique_frames = listen_landmark[\"frame\"].nunique()\n    unique_types = listen_landmark[\"type\"].nunique()\n\n    print(f\"This file has total of {unique_frames} unique frames\")\n    print(\n        f\"This file has total of {unique_types} unique types which contain {listen_landmark['type'].unique()}\"\n    )\n    print(\"------------------------------------------------------\")\n    if i == 10:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:53.398631Z","iopub.execute_input":"2023-04-13T13:07:53.399664Z","iopub.status.idle":"2023-04-13T13:07:53.680275Z","shell.execute_reply.started":"2023-04-13T13:07:53.399605Z","shell.execute_reply":"2023-04-13T13:07:53.678892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# create a metadata for training dataset","metadata":{}},{"cell_type":"code","source":"N_PARQUETS_TO_READ = 100_0  # So we don't have to load all 95k\n\ncombined_meta = {}\nfor i, d in tqdm(train.iterrows(), total=len(train)):\n    file_path = d[\"path\"]\n    example_landmark = pd.read_parquet(f\"{BASE_DIR}/{file_path}\")\n    # Get the number of landmarks with x,y,z data per type\n    meta = (\n        example_landmark.dropna(subset=[\"x\", \"y\", \"z\"])[\"type\"].value_counts().to_dict()\n    )\n    meta[\"frames\"] = example_landmark[\"frame\"].nunique()\n    xyz_meta = (\n        example_landmark.agg(\n            {\n                \"x\": [\"min\", \"max\", \"mean\"],\n                \"y\": [\"min\", \"max\", \"mean\"],\n                \"z\": [\"min\", \"max\", \"mean\"],\n            }\n        )\n        .unstack()\n        .to_dict()\n    )\n\n    for key in xyz_meta.keys():\n        new_key = key[0] + \"_\" + key[1]\n        meta[new_key] = xyz_meta[key]\n    combined_meta[file_path] = meta\n    if i >= N_PARQUETS_TO_READ:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:07:53.681846Z","iopub.execute_input":"2023-04-13T13:07:53.683355Z","iopub.status.idle":"2023-04-13T13:08:29.819978Z","shell.execute_reply.started":"2023-04-13T13:07:53.683301Z","shell.execute_reply":"2023-04-13T13:08:29.818542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(combined_meta).T.reset_index().rename(columns={\"index\": \"path\"})","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:29.821743Z","iopub.execute_input":"2023-04-13T13:08:29.822245Z","iopub.status.idle":"2023-04-13T13:08:30.030681Z","shell.execute_reply.started":"2023-04-13T13:08:29.822194Z","shell.execute_reply":"2023-04-13T13:08:30.029560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_with_meta = train.merge(\n    pd.DataFrame(combined_meta).T.reset_index().rename(columns={\"index\": \"path\"}),\n    how=\"left\",\n)\ntrain_with_meta.to_parquet(\"train_with_meta.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.031936Z","iopub.execute_input":"2023-04-13T13:08:30.032248Z","iopub.status.idle":"2023-04-13T13:08:30.238286Z","shell.execute_reply.started":"2023-04-13T13:08:30.032219Z","shell.execute_reply":"2023-04-13T13:08:30.237298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_with_meta","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.240473Z","iopub.execute_input":"2023-04-13T13:08:30.240954Z","iopub.status.idle":"2023-04-13T13:08:30.376419Z","shell.execute_reply.started":"2023-04-13T13:08:30.240905Z","shell.execute_reply":"2023-04-13T13:08:30.375272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# what are the most frequent types of landmarks provided\n    - Face has a lot of datapoints because mediapipe provides 468 3D datapoints per frame","metadata":{}},{"cell_type":"code","source":"train_with_meta[[\"face\", \"pose\", \"left_hand\", \"right_hand\"]].sum().sort_values().plot(\n    kind=\"barh\", color=\"green\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.377807Z","iopub.execute_input":"2023-04-13T13:08:30.378121Z","iopub.status.idle":"2023-04-13T13:08:30.616178Z","shell.execute_reply.started":"2023-04-13T13:08:30.378091Z","shell.execute_reply":"2023-04-13T13:08:30.614975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(\n    train_with_meta.query(\"index<1000\").fillna(0)[\n        [\"face\", \"left_hand\", \"pose\", \"right_hand\"]\n    ]\n    > 0\n).mean().plot(kind=\"barh\", title=\"Keypoints with data\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.617440Z","iopub.execute_input":"2023-04-13T13:08:30.617777Z","iopub.status.idle":"2023-04-13T13:08:30.786544Z","shell.execute_reply.started":"2023-04-13T13:08:30.617744Z","shell.execute_reply":"2023-04-13T13:08:30.785634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_shhh = train.query('sign==\"shhh\"')[\"path\"].values[0]\nshhh_landmark = pd.read_parquet(f\"{BASE_DIR}{first_shhh}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.787475Z","iopub.execute_input":"2023-04-13T13:08:30.787771Z","iopub.status.idle":"2023-04-13T13:08:30.811977Z","shell.execute_reply.started":"2023-04-13T13:08:30.787741Z","shell.execute_reply":"2023-04-13T13:08:30.811099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shhh_landmark.query(\"frame==25\")[\"type\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.813103Z","iopub.execute_input":"2023-04-13T13:08:30.813636Z","iopub.status.idle":"2023-04-13T13:08:30.832131Z","shell.execute_reply.started":"2023-04-13T13:08:30.813601Z","shell.execute_reply":"2023-04-13T13:08:30.830807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_landmark[\"no_xyz\"] = example_landmark[\"x\"].isna()\nexample_landmark.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.833799Z","iopub.execute_input":"2023-04-13T13:08:30.834773Z","iopub.status.idle":"2023-04-13T13:08:30.857594Z","shell.execute_reply.started":"2023-04-13T13:08:30.834721Z","shell.execute_reply":"2023-04-13T13:08:30.856343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_landmark.groupby(\"frame\")[\"no_xyz\"].sum().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:30.860557Z","iopub.execute_input":"2023-04-13T13:08:30.861261Z","iopub.status.idle":"2023-04-13T13:08:31.262909Z","shell.execute_reply.started":"2023-04-13T13:08:30.861219Z","shell.execute_reply":"2023-04-13T13:08:31.261690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"frame 17 and 19 don't have missing x","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\nexample_frame = example_landmark.query(\"frame==17\")\npx.scatter_3d(example_frame, x=\"x\", y=\"y\", z=\"z\", color=\"type\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:31.265462Z","iopub.execute_input":"2023-04-13T13:08:31.266648Z","iopub.status.idle":"2023-04-13T13:08:31.454250Z","shell.execute_reply.started":"2023-04-13T13:08:31.266610Z","shell.execute_reply":"2023-04-13T13:08:31.453326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_landmark[\"y_\"] = example_landmark[\"y\"] * -1\nexample_frame = example_landmark.query('frame == 17 and type == \"face\"')\npx.scatter(example_frame, x=\"x\", y=\"y_\")","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:31.459025Z","iopub.execute_input":"2023-04-13T13:08:31.459751Z","iopub.status.idle":"2023-04-13T13:08:31.548250Z","shell.execute_reply.started":"2023-04-13T13:08:31.459709Z","shell.execute_reply":"2023-04-13T13:08:31.547042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mediapipe --quiet","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:31.549616Z","iopub.execute_input":"2023-04-13T13:08:31.549958Z","iopub.status.idle":"2023-04-13T13:08:43.079109Z","shell.execute_reply.started":"2023-04-13T13:08:31.549926Z","shell.execute_reply":"2023-04-13T13:08:43.077348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mediapipe as mp\n\nmp_hands = mp.solutions.hands\n\n\nexample_landmark[\"y_\"] = example_landmark[\"y\"] * -1\n\nfig, ax = plt.subplots(figsize=(5, 5))\n\nfor hand in [\"left_hand\", \"right_hand\"]:\n    example_hand = example_landmark.query(\"frame == 17 and type == @hand\")\n\n    ax.scatter(example_hand[\"x\"], example_hand[\"y_\"])\n\n    for connection in mp_hands.HAND_CONNECTIONS:\n        point_a = connection[0]\n        point_b = connection[1]\n        x1, y1 = example_hand.query(\"landmark_index == @point_a\")[[\"x\", \"y_\"]].values[0]\n        x2, y2 = example_hand.query(\"landmark_index == @point_b\")[[\"x\", \"y_\"]].values[0]\n        plt.plot([x1, x2], [y1, y2], color=\"purple\")\nax.set_title(\"Shhh - Hands Data\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:43.082135Z","iopub.execute_input":"2023-04-13T13:08:43.082563Z","iopub.status.idle":"2023-04-13T13:08:43.661508Z","shell.execute_reply.started":"2023-04-13T13:08:43.082522Z","shell.execute_reply":"2023-04-13T13:08:43.660253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:14:28.574239Z","iopub.execute_input":"2023-04-13T13:14:28.574703Z","iopub.status.idle":"2023-04-13T13:14:28.582833Z","shell.execute_reply.started":"2023-04-13T13:14:28.574660Z","shell.execute_reply":"2023-04-13T13:14:28.581442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ROWS_PER_FRAME = 543  # number of landmarks per frame\n\n# def load_relevant_data_subset(pq_path):\n#     data_columns = ['x', 'y', 'z']\n#     data = pd.read_parquet(pq_path, columns=data_columns)\n#     n_frames = int(len(data) / ROWS_PER_FRAME)\n#     data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n#     return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:43.663415Z","iopub.execute_input":"2023-04-13T13:08:43.663889Z","iopub.status.idle":"2023-04-13T13:08:43.669944Z","shell.execute_reply.started":"2023-04-13T13:08:43.663852Z","shell.execute_reply":"2023-04-13T13:08:43.668653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tflite_runtime.interpreter as tflite\n# def run_model(model_path):\n#     interpreter = tflite.Interpreter(model_path)\n\n#     found_signatures = list(interpreter.get_signature_list().keys())\n\n#     if REQUIRED_SIGNATURE not in found_signatures:\n#         raise KernelEvalException('Required input signature not found.')\n\n#     prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n#     output = prediction_fn(inputs=frames)\n#     sign = np.argmax(output[\"outputs\"])","metadata":{"execution":{"iopub.status.busy":"2023-04-13T13:08:43.671784Z","iopub.execute_input":"2023-04-13T13:08:43.672095Z","iopub.status.idle":"2023-04-13T13:08:43.682203Z","shell.execute_reply.started":"2023-04-13T13:08:43.672065Z","shell.execute_reply":"2023-04-13T13:08:43.681177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}