{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"},{"sourceId":5770540,"sourceType":"datasetVersion","datasetId":3250208},{"sourceId":7005376,"sourceType":"datasetVersion","datasetId":4025509}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install opencv-python mediapipe numpy pyarrow","metadata":{"execution":{"iopub.status.busy":"2023-11-20T10:05:03.507854Z","iopub.execute_input":"2023-11-20T10:05:03.508290Z","iopub.status.idle":"2023-11-20T10:05:21.063630Z","shell.execute_reply.started":"2023-11-20T10:05:03.508255Z","shell.execute_reply":"2023-11-20T10:05:21.062380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-11-20T10:05:21.065955Z","iopub.execute_input":"2023-11-20T10:05:21.066433Z","iopub.status.idle":"2023-11-20T10:05:21.071487Z","shell.execute_reply.started":"2023-11-20T10:05:21.066391Z","shell.execute_reply":"2023-11-20T10:05:21.070741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annot = pd.DataFrame(columns=['path','user_id','sign'])","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:25:03.852574Z","iopub.execute_input":"2023-11-20T07:25:03.853093Z","iopub.status.idle":"2023-11-20T07:25:03.869564Z","shell.execute_reply.started":"2023-11-20T07:25:03.853034Z","shell.execute_reply":"2023-11-20T07:25:03.868527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annot2 = pd.read_csv('/kaggle/input/slovo/annotations.csv',on_bad_lines='skip')","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:25:06.268713Z","iopub.execute_input":"2023-11-20T07:25:06.269740Z","iopub.status.idle":"2023-11-20T07:25:06.382470Z","shell.execute_reply.started":"2023-11-20T07:25:06.269678Z","shell.execute_reply":"2023-11-20T07:25:06.381374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# columns = list(annot2.columns)[0].split(\"\\t\")\n# new_annot = pd.DataFrame(columns = columns)\n# for index, row in annot2.iterrows():\n#     dic = {}\n#     values = list(row.values[0].split(\"\\t\"))\n#     dic  = {'attachment_id': values[0],'text': values[1],'user_id': values[2],'height': values[3],'width': values[4],'length': values[5],'train': values[6],'begin': values[7],'end': values[8]}\n#     new_annot =  new_annot.append(dic, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:25:09.836911Z","iopub.execute_input":"2023-11-20T07:25:09.837391Z","iopub.status.idle":"2023-11-20T07:26:15.384930Z","shell.execute_reply.started":"2023-11-20T07:25:09.837351Z","shell.execute_reply":"2023-11-20T07:26:15.383925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# new_annot[new_annot.attachment_id=='8f8056fe-ceb5-4324-a5ca-6d8e95184692.parquet']","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:10:48.037125Z","iopub.execute_input":"2023-11-19T19:10:48.037593Z","iopub.status.idle":"2023-11-19T19:10:48.056364Z","shell.execute_reply.started":"2023-11-19T19:10:48.037557Z","shell.execute_reply":"2023-11-19T19:10:48.053914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# new_annot","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:26:15.386869Z","iopub.execute_input":"2023-11-20T07:26:15.387611Z","iopub.status.idle":"2023-11-20T07:26:15.424602Z","shell.execute_reply.started":"2023-11-20T07:26:15.387572Z","shell.execute_reply":"2023-11-20T07:26:15.423394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = new_annot[new_annot.train=='True']","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:05:35.339039Z","iopub.execute_input":"2023-11-19T19:05:35.339718Z","iopub.status.idle":"2023-11-19T19:05:35.364336Z","shell.execute_reply.started":"2023-11-19T19:05:35.339666Z","shell.execute_reply":"2023-11-19T19:05:35.362361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path = []\n# for index, row in train.iterrows():\n#     path_text = \"output/train/\"+ row['attachment_id']\n#     path.append(path_text)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:05:35.365822Z","iopub.execute_input":"2023-11-19T19:05:35.366155Z","iopub.status.idle":"2023-11-19T19:05:36.218268Z","shell.execute_reply.started":"2023-11-19T19:05:35.366126Z","shell.execute_reply":"2023-11-19T19:05:36.215076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annot['path'] = path\n# annot['user_id'] = train['user_id']\n# annot['sign'] = train['text']","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:05:36.221538Z","iopub.execute_input":"2023-11-19T19:05:36.222302Z","iopub.status.idle":"2023-11-19T19:05:36.239712Z","shell.execute_reply.started":"2023-11-19T19:05:36.222245Z","shell.execute_reply":"2023-11-19T19:05:36.238051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test = new_annot[new_annot.train=='False']","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:05:36.241341Z","iopub.execute_input":"2023-11-19T19:05:36.241766Z","iopub.status.idle":"2023-11-19T19:05:36.257153Z","shell.execute_reply.started":"2023-11-19T19:05:36.241731Z","shell.execute_reply":"2023-11-19T19:05:36.255153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annot","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:17:24.992290Z","iopub.execute_input":"2023-11-12T12:17:24.992711Z","iopub.status.idle":"2023-11-12T12:17:25.010702Z","shell.execute_reply.started":"2023-11-12T12:17:24.992678Z","shell.execute_reply":"2023-11-12T12:17:25.009730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annot.to_csv('train.csv', index=False)\n# test.to_csv('test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:05:50.408693Z","iopub.execute_input":"2023-11-19T19:05:50.409091Z","iopub.status.idle":"2023-11-19T19:05:50.491141Z","shell.execute_reply.started":"2023-11-19T19:05:50.409062Z","shell.execute_reply":"2023-11-19T19:05:50.490392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annot[annot.path=='output/train/8f8056fe-ceb5-4324-a5ca-6d8e95184692']","metadata":{"execution":{"iopub.status.busy":"2023-11-19T19:48:49.987190Z","iopub.execute_input":"2023-11-19T19:48:49.987593Z","iopub.status.idle":"2023-11-19T19:48:50.001888Z","shell.execute_reply.started":"2023-11-19T19:48:49.987563Z","shell.execute_reply":"2023-11-19T19:48:50.000703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport mediapipe as mp\nimport pandas as pd\nimport os\n\n# Load MediaPipe Holistic model\nmp_drawing = mp.solutions.drawing_utils\nmp_holistic = mp.solutions.holistic\n\n# Define input dataset directory and output directory for parquet files\ndataset_dir = ['/kaggle/input/slovo/slovo/train','/kaggle/input/slovo/slovo/test']\noutput_dir = ['/kaggle/working/output/train','/kaggle/working/output/test']\n\n# Create the output directory if it doesn't exist\nos.makedirs('/kaggle/working/output', exist_ok=True)\nos.makedirs(output_dir[0], exist_ok=True)\nos.makedirs(output_dir[1], exist_ok=True)\n\n# Initialize MediaPipe Holistic model\nholistic = mp_holistic.Holistic()\n\ndef process_video(video_path, output_path):\n    # Initialize video reader\n    video_reader = cv2.VideoCapture(video_path)\n\n    frames = []\n    frame_index = 0\n\n    while video_reader.isOpened():\n        # Read a frame from the video\n        success, frame = video_reader.read()\n        if not success:\n            break\n\n        # Convert the frame to RGB for MediaPipe\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n\n        # Run holistic estimation on the frame\n        results = holistic.process(frame_rgb)\n\n        # Extract face landmarks\n        if results.face_landmarks:\n            for i, landmark in enumerate(results.face_landmarks.landmark):\n                frames.append([frame_index, f'{frame_index}-face-{i}', 'face', i, landmark.x, landmark.y, landmark.z])\n\n        # Extract hand landmarks\n        if results.left_hand_landmarks:\n            for i, landmark in enumerate(results.left_hand_landmarks.landmark):\n                frames.append([frame_index, f'{frame_index}-hand-{i}', 'left_hand', i, landmark.x, landmark.y, landmark.z])\n\n        if results.right_hand_landmarks:\n            for i, landmark in enumerate(results.right_hand_landmarks.landmark):\n                frames.append([frame_index, f'{frame_index}-hand-{i}', 'right_hand', i, landmark.x, landmark.y, landmark.z])\n        if results.pose_landmarks:\n            for i, landmark in enumerate(results.pose_landmarks.landmark):\n                frames.append([frame_index, f'{frame_index}-hand-{i}', 'pose', i, landmark.x, landmark.y, landmark.z])\n\n        frame_index += 1\n\n    # Release resources\n    video_reader.release()\n\n    # Create a dataframe from the extracted landmarks\n    df = pd.DataFrame(frames, columns=['frame', 'row_id', 'type', 'landmark_index', 'x', 'y', 'z'])\n\n    # Save the dataframe to a parquet file\n    df.to_parquet(output_path)\ncount = 0\nfor i in range(2):\n    for filename in os.listdir(dataset_dir[i]):\n        if filename.endswith('.mp4') and count<9000:\n            \n            if os.path.exists(\"/kaggle/working/file.txt\"):\n                file = open('file.txt', 'w')\n                file.write(str(count))\n                file.close()\n            else:\n                with open('file.txt', 'a') as f:\n                   f.write(str(count))\n            video_path = os.path.join(dataset_dir[i], filename)\n            output_path = os.path.join(output_dir[i], f'{os.path.splitext(filename)[0]}.parquet')\n    \n            # Process the video and save landmarks to parquet file\n            process_video(video_path, output_path)\n        count+=1","metadata":{"execution":{"iopub.status.busy":"2023-11-20T10:05:43.451698Z","iopub.execute_input":"2023-11-20T10:05:43.453042Z","iopub.status.idle":"2023-11-20T10:06:48.544142Z","shell.execute_reply.started":"2023-11-20T10:05:43.452981Z","shell.execute_reply":"2023-11-20T10:06:48.542666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import cv2\n# import mediapipe as mp\n# import pandas as pd\n# import os\n\n# # Load MediaPipe Holistic model\n# mp_holistic = mp.solutions.holistic\n\n# # Define input dataset directory and output directory for parquet files\n# dataset_dir = ['/kaggle/input/slovo/slovo/train', '/kaggle/input/slovo/slovo/test']\n# output_dir = ['/kaggle/working/output/train', '/kaggle/working/output/test']\n\n# # Create the output directory if it doesn't exist\n# os.makedirs('/kaggle/working/output', exist_ok=True)\n# os.makedirs(output_dir[0], exist_ok=True)\n# os.makedirs(output_dir[1], exist_ok=True)\n\n# # Initialize MediaPipe Holistic model\n# holistic = mp_holistic.Holistic()\n\n# def process_video(video_path, output_path):\n#     # Initialize video reader\n#     video_reader = cv2.VideoCapture(video_path)\n\n#     frames = []\n#     frame_index = 0\n\n#     while video_reader.isOpened():\n#         # Read a frame from the video\n#         success, frame = video_reader.read()\n#         if not success:\n#             break\n\n#         # Convert the frame to RGB for MediaPipe\n#         frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n\n#         # Run holistic estimation on the frame\n#         results = holistic.process(frame_rgb)\n\n#         # Extract face, left hand, and right hand landmarks\n#         for landmark_type in ['face_landmarks', 'left_hand_landmarks', 'right_hand_landmarks']:\n#             landmarks = getattr(results, landmark_type, None)\n#             if landmarks:\n#                 for i, landmark in enumerate(landmarks.landmark):\n#                     # Add data to the DataFrame\n#                     row_id = f\"{frame_index}-{landmark_type.split('_')[0]}-{i}\"\n#                     frames.append([frame_index, row_id, landmark_type.split('_')[0], i, landmark.x, landmark.y, landmark.z if landmark.HasField('z') else None])\n\n#                 # Check for missing landmarks and add rows with 'missing' for landmark_index\n#                 missing_landmarks = set(range(len(landmarks.landmark))) - set([i for i, _ in enumerate(landmarks.landmark)])\n#                 for missing_index in missing_landmarks:\n#                     row_id_missing = f\"{frame_index}-{landmark_type.split('_')[0]}-{missing_index}\"\n#                     frames.append([frame_index, row_id_missing, landmark_type.split('_')[0], 'missing', None, None, None])\n\n#         frame_index += 1\n\n#     # Release resources\n#     video_reader.release()\n\n#     # Create a DataFrame from the extracted landmarks\n#     columns = ['frame', 'row_id', 'type', 'landmark_index', 'x', 'y', 'z']\n#     df = pd.DataFrame(frames, columns=columns)\n\n#     # Save the DataFrame to a parquet file\n#     df.to_parquet(output_path)\n\n# count = 0\n# for i in range(2):\n#     for filename in os.listdir(dataset_dir[i]):\n#         if count == 4:\n#             break\n#         if filename.endswith('.mp4'):\n#             count += 1\n#             if os.path.exists(\"/kaggle/working/file.txt\"):\n#                 file = open('file.txt', 'w')\n#                 file.write(str(count))\n#                 file.close()\n#             else:\n#                 with open('file.txt', 'a') as f:\n#                     f.write(str(count))\n#             video_path = os.path.join(dataset_dir[i], filename)\n#             output_path = os.path.join(output_dir[i], f'{os.path.splitext(filename)[0]}.parquet')\n\n#             # Process the video and save landmarks to parquet file\n#             process_video(video_path, output_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-18T08:36:06.719090Z","iopub.execute_input":"2023-11-18T08:36:06.719617Z","iopub.status.idle":"2023-11-18T08:36:26.156821Z","shell.execute_reply.started":"2023-11-18T08:36:06.719583Z","shell.execute_reply":"2023-11-18T08:36:26.155562Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# # dt = pd.read_parquet('/kaggle/input/asl-signs/train_landmark_files/16069/100015657.parquet')\n# # dt","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:11:27.805807Z","iopub.execute_input":"2023-11-20T07:11:27.806259Z","iopub.status.idle":"2023-11-20T07:11:27.838204Z","shell.execute_reply.started":"2023-11-20T07:11:27.806228Z","shell.execute_reply":"2023-11-20T07:11:27.836667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# names_of_files = []\n# for filename in os.listdir('/kaggle/input/mediapipe-slovo/output/output/train'):\n#     name = os.path.splitext(filename)[0]\n#     names_of_files.append(name)","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:28:14.647249Z","iopub.execute_input":"2023-11-20T07:28:14.647700Z","iopub.status.idle":"2023-11-20T07:28:15.315875Z","shell.execute_reply.started":"2023-11-20T07:28:14.647667Z","shell.execute_reply":"2023-11-20T07:28:15.314589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# names_of_files","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:28:23.558080Z","iopub.execute_input":"2023-11-20T07:28:23.558726Z","iopub.status.idle":"2023-11-20T07:28:23.564908Z","shell.execute_reply.started":"2023-11-20T07:28:23.558686Z","shell.execute_reply":"2023-11-20T07:28:23.563795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_filtered = new_annot[new_annot['attachment_id'].isin(names_of_files)]","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:33:00.961881Z","iopub.execute_input":"2023-11-20T07:33:00.962331Z","iopub.status.idle":"2023-11-20T07:33:00.990844Z","shell.execute_reply.started":"2023-11-20T07:33:00.962302Z","shell.execute_reply":"2023-11-20T07:33:00.989155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_filtered","metadata":{"execution":{"iopub.status.busy":"2023-11-20T07:33:07.079513Z","iopub.execute_input":"2023-11-20T07:33:07.079958Z","iopub.status.idle":"2023-11-20T07:33:07.101766Z","shell.execute_reply.started":"2023-11-20T07:33:07.079925Z","shell.execute_reply":"2023-11-20T07:33:07.100419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}