{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q flatbuffers 2> /dev/null\n!pip install -q mediapipe 2> /dev/null","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-25T04:10:01.483614Z","iopub.execute_input":"2023-03-25T04:10:01.483973Z","iopub.status.idle":"2023-03-25T04:10:19.668184Z","shell.execute_reply.started":"2023-03-25T04:10:01.483944Z","shell.execute_reply":"2023-03-25T04:10:19.666288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport mediapipe as mp\nimport matplotlib.pyplot as plt\n\nfrom matplotlib import animation\nfrom pathlib import Path\nimport IPython\nfrom IPython import display\nfrom IPython.display import HTML\n\nimport mediapipe as mp\nfrom mediapipe.framework.formats import landmark_pb2\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:19.671407Z","iopub.execute_input":"2023-03-25T04:10:19.671680Z","iopub.status.idle":"2023-03-25T04:10:19.680384Z","shell.execute_reply.started":"2023-03-25T04:10:19.671651Z","shell.execute_reply":"2023-03-25T04:10:19.678638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Cfg:\n    RANDOM_STATE = 2023\n    INPUT_ROOT = Path('/kaggle/input/asl-signs/')\n    OUTPUT_ROOT = Path('kaggle/working')\n    INDEX_MAP_FILE = INPUT_ROOT / 'sign_to_prediction_index_map.json'\n    TRAN_FILE = INPUT_ROOT / 'train.csv'\n    INDEX = 'sequence_id'\n    ROW_ID = 'row_id'","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:19.683115Z","iopub.execute_input":"2023-03-25T04:10:19.683566Z","iopub.status.idle":"2023-03-25T04:10:19.699827Z","shell.execute_reply.started":"2023-03-25T04:10:19.683534Z","shell.execute_reply":"2023-03-25T04:10:19.698282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_index_map(file_path=Cfg.INDEX_MAP_FILE):\n    \"\"\"Reads the sign to predict as json file.\"\"\"\n    with open(file_path, \"r\") as f:\n        result = json.load(f)\n    return result    \n\ndef read_train(file_path=Cfg.TRAN_FILE):\n    \"\"\"Reads the train csv as pandas data frame.\"\"\"\n    return pd.read_csv(file_path).set_index(Cfg.INDEX)\n\ndef read_landmark_data_by_path(file_path, input_root=Cfg.INPUT_ROOT):\n    \"\"\"Reads landmak data by the given file path.\"\"\"\n    data = pd.read_parquet(input_root / file_path)\n    return data.set_index(Cfg.ROW_ID)\n\ndef read_landmark_data_by_id(sequence_id, train_data):\n    \"\"\"Reads the landmark data by the given sequence id.\"\"\"\n    file_path = train_data.loc[sequence_id]['path']\n    return read_landmark_data_by_path(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:19.702666Z","iopub.execute_input":"2023-03-25T04:10:19.703040Z","iopub.status.idle":"2023-03-25T04:10:19.711712Z","shell.execute_reply.started":"2023-03-25T04:10:19.703007Z","shell.execute_reply":"2023-03-25T04:10:19.710750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = read_train()\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:19.712645Z","iopub.execute_input":"2023-03-25T04:10:19.712931Z","iopub.status.idle":"2023-03-25T04:10:19.836419Z","shell.execute_reply.started":"2023-03-25T04:10:19.712904Z","shell.execute_reply":"2023-03-25T04:10:19.835428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequence_id = 1000106739\ndata = read_landmark_data_by_id(sequence_id, train_data)\nprint(data)","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:19.837308Z","iopub.execute_input":"2023-03-25T04:10:19.837641Z","iopub.status.idle":"2023-03-25T04:10:19.867652Z","shell.execute_reply.started":"2023-03-25T04:10:19.837604Z","shell.execute_reply":"2023-03-25T04:10:19.866655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHANGE TO THE SIGN YOU WANT TO CREATE DATASET FOR\nsign = \"blue\"\n# train_data.query('sign == \"listen\"')\nnum_of_files = len(train_data.query(f'sign == \"{sign}\"'))\nnum_of_files","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:35.424332Z","iopub.execute_input":"2023-03-25T04:10:35.424650Z","iopub.status.idle":"2023-03-25T04:10:35.436405Z","shell.execute_reply.started":"2023-03-25T04:10:35.424622Z","shell.execute_reply":"2023-03-25T04:10:35.435273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(sign):\n    print(\"making directory\")\n    os.makedirs(sign)","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:10:38.626019Z","iopub.execute_input":"2023-03-25T04:10:38.626427Z","iopub.status.idle":"2023-03-25T04:10:38.633196Z","shell.execute_reply.started":"2023-03-25T04:10:38.626398Z","shell.execute_reply":"2023-03-25T04:10:38.631966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# counter for videos - for first video it should be one     \ncounter = 0\n\nfor video in range(0,num_of_files):\n    link_to_video = f\"/kaggle/input/asl-signs/{train_data.query(f'sign == @sign')['path'].values[video]}\"\n\n\n    data = pd.read_parquet(link_to_video)\n\n    unique_frames = data[\"frame\"].nunique()\n    unique_types = data[\"type\"].nunique()\n    types_in_video = data[\"type\"].unique()\n    #     print(\n    #         f\"The file has {unique_frames} unique frames and {unique_types} unique types: {types_in_video}\"\n    #     )\n    \n    # counter for frames in video - for first frame it should be one     \n    frame_counter = 0\n    \n    for frame, group in data.groupby('frame'):\n        \n#         print(f\"Data for frame {frame}:\")\n\n        #     print(len(group.query('type == \"pose\"')))\n        body_parts = ['pose','face', 'left_hand', 'right_hand']\n        concate_frame_data = np.empty(0)\n\n        for part in body_parts: \n            frame_data = group.query(f\"type == '{part}'\")\n\n            individual_frame = np.empty((0, 3))\n            for index, row in frame_data.iterrows():\n                individual_frame = np.append(individual_frame, [[row.x, row.y, row.z]], axis=0)\n\n            flatten_data = individual_frame.flatten()\n\n            if np.any(np.isnan(flatten_data)) == True:\n                flatten_data = np.zeros(flatten_data.shape)\n\n            concate_frame_data = np.append(concate_frame_data,flatten_data)\n\n        #         print(flatten_data.shape,np.any(np.isnan(flatten_data)),flatten_data)\n        #         print(concate_frame_data,concate_frame_data.shape)\n        if not os.path.exists(f\"/kaggle/working/{sign}/{counter}\"):\n#             print(f\"/kaggle/working/{sign}/{counter} path doesn't exist\")\n            os.makedirs(f\"/kaggle/working/{sign}/{counter}\")\n        \n        \n        path_to_save = f\"/kaggle/working/{sign}/{counter}/{frame_counter}\" \n#         print(path_to_save)\n        np.save(path_to_save,concate_frame_data)\n        frame_counter += 1\n        \n    counter += 1\n#     if counter == 4:\n#         break","metadata":{"execution":{"iopub.status.busy":"2023-03-25T04:11:08.067917Z","iopub.execute_input":"2023-03-25T04:11:08.068273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after creation of folder don't view on kaggle, just add the folder name here.\n!zip -r file.zip /kaggle/working/black","metadata":{"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove a existing folder. DON'T RUN, ONLY RUN WHEN YOU WANT TO DELETE A FOLDER\nimport shutil\nshutil.rmtree(\"/kaggle/working/listen\")","metadata":{"execution":{"iopub.status.busy":"2023-03-25T03:33:22.759033Z","iopub.execute_input":"2023-03-25T03:33:22.759464Z","iopub.status.idle":"2023-03-25T03:33:22.857351Z","shell.execute_reply.started":"2023-03-25T03:33:22.759428Z","shell.execute_reply":"2023-03-25T03:33:22.856126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_frames = data[\"frame\"].nunique()\nunique_types = data[\"type\"].nunique()\ntypes_in_video = data[\"type\"].unique()\nprint(\n    f\"The file has {unique_frames} unique frames and {unique_types} unique types: {types_in_video}\"\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(num_of_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for frame, group in data.groupby('frame'):\n    print(f\"Data for frame {frame}:\")\n    \n    #     print(len(group.query('type == \"pose\"')))\n    body_parts = ['pose','face', 'left_hand', 'right_hand']\n    concate_frame_data = np.empty(0)\n    \n    for part in body_parts: \n        frame_data = group.query(f\"type == '{part}'\")\n        \n        individual_frame = np.empty((0, 3))\n        for index, row in frame_data.iterrows():\n            individual_frame = np.append(individual_frame, [[row.x, row.y, row.z]], axis=0)\n\n        flatten_data = individual_frame.flatten()\n\n        if np.any(np.isnan(flatten_data)) == True:\n            flatten_data = np.zeros(flatten_data.shape)\n        \n        concate_frame_data = np.append(concate_frame_data,flatten_data)\n\n#         print(flatten_data.shape,np.any(np.isnan(flatten_data)),flatten_data)\n    print(concate_frame_data,concate_frame_data.shape)\n    \n    np.save(\"0\",concate_frame_data)\n    \n    break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}