{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"},{"sourceId":2632847,"sourceType":"datasetVersion","datasetId":1589971}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'  # Suppress all TF messages except errors\nos.environ['TF_ENABLE_ONEDNN_OPTS'] = '0'  \n# Disable oneDNN optimizations warnings\nimport sys\nimport warnings\nimport logging\nfrom contextlib import redirect_stderr\nimport io\n\n# HARDCORE SUPPRESSION - Put this at the VERY TOP of your script\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nos.environ['TF_ENABLE_ONEDNN_OPTS'] = '0'\nos.environ['CUDA_VISIBLE_DEVICES'] = ''  # Force CPU only to reduce warnings\nos.environ['TF_FORCE_GPU_ALLOW_GROWTH'] = 'true'\n\n# Suppress all warnings\nwarnings.filterwarnings('ignore')\nlogging.getLogger().setLevel(logging.ERROR)\n\n# Redirect stderr to suppress C++ level warnings\nclass SuppressStderr:\n    def __enter__(self):\n        self._original_stderr = sys.stderr\n        sys.stderr = open(os.devnull, 'w')\n        return self\n\n    def __exit__(self, exc_type, exc_val, exc_tb):\n        sys.stderr.close()\n        sys.stderr = self._original_stderr\n\n# Import TensorFlow AFTER setting environment variables\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR')\ntf.compat.v1.logging.set_verbosity(tf.compat.v1.logging.ERROR)\n\n# Suppress ABSL\ntry:\n    import absl.logging\n    absl.logging.set_verbosity(absl.logging.ERROR)\nexcept:\n    pass\nimport numpy as np # linear algebra\n\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-06-14T17:14:35.214747Z","iopub.execute_input":"2025-06-14T17:14:35.215327Z","iopub.status.idle":"2025-06-14T17:14:35.223654Z","shell.execute_reply.started":"2025-06-14T17:14:35.215297Z","shell.execute_reply":"2025-06-14T17:14:35.222720Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%capture \n!pip install mediapipe","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:04:47.390668Z","iopub.execute_input":"2025-06-14T17:04:47.391277Z","iopub.status.idle":"2025-06-14T17:04:55.646956Z","shell.execute_reply.started":"2025-06-14T17:04:47.391244Z","shell.execute_reply":"2025-06-14T17:04:55.645712Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport mediapipe as mp\nmp_drawing = mp.solutions.drawing_utils\nmp_drawing_styles = mp.solutions.drawing_styles\nmp_holistic = mp.solutions.holistic\nfrom IPython.display import Image, display\nimport matplotlib.pyplot as plt\ndef transform(path , start_frame , end_frame , fps):\n    frame_number = 0\n    frame = []\n    type_ = []\n    index = []\n    x = []\n    y = []\n    z = []\n    \n    cap = cv2.VideoCapture(path)\n    cap.set(cv2.CAP_PROP_FPS, fps)\n    with mp_holistic.Holistic(min_detection_confidence=0.5,min_tracking_confidence=0.5) as holistic:\n        while cap.isOpened():\n            success, image = cap.read()\n            if not success:\n                break\n            frame_number += 1\n            if frame_number < start_frame:\n                continue\n            if end_frame != -1 and frame_number > end_frame:\n                break\n            image.flags.writeable = False\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            results = holistic.process(image)\n            #face\n            if(results.face_landmarks is None):\n                for i in range(478):\n                    frame.append(frame_number)\n                    type_.append(\"face\")\n                    index.append(ind)\n                    x.append(0)\n                    y.append(0)\n                    z.append(0)\n            else:\n                for ind,val in enumerate(results.face_landmarks.landmark):\n                    frame.append(frame_number)\n                    type_.append(\"face\")\n                    index.append(ind)\n                    x.append(val.x)\n                    y.append(val.y)\n                    z.append(val.z)\n            #pose\n            if(results.pose_landmarks is None):\n                for i in range(32):\n                    frame.append(frame_number)\n                    type_.append(\"pose\")\n                    index.append(ind)\n                    x.append(0)\n                    y.append(0)\n                    z.append(0)\n            else:\n                for ind,val in enumerate(results.pose_landmarks.landmark):\n                    frame.append(frame_number)\n                    type_.append(\"pose\")\n                    index.append(ind)\n                    x.append(val.x)\n                    y.append(val.y)\n                    z.append(val.z)\n            #left hand\n            if(results.left_hand_landmarks is None):\n                for i in range(20):\n                    frame.append(frame_number)\n                    type_.append(\"left_hand\")\n                    index.append(ind)\n                    x.append(0)\n                    y.append(0)\n                    z.append(0)\n            else:\n                for ind,val in enumerate(results.left_hand_landmarks.landmark):\n                    frame.append(frame_number)\n                    type_.append(\"left_hand\")\n                    index.append(ind)\n                    x.append(val.x)\n                    y.append(val.y)\n                    z.append(val.z)\n            #right hand\n            if(results.right_hand_landmarks is None):\n                for i in range(20):\n                    frame.append(frame_number)\n                    type_.append(\"right_hand\")\n                    index.append(ind)\n                    x.append(0)\n                    y.append(0)\n                    z.append(0)\n            else:\n                for ind,val in enumerate(results.right_hand_landmarks.landmark):\n                    frame.append(frame_number)\n                    type_.append(\"right_hand\")\n                    index.append(ind)\n                    x.append(val.x)\n                    y.append(val.y)\n                    z.append(val.z)\n            \n    return pd.DataFrame({\n        \"frame\" : frame,\n        \"type\"  : type_,\n        \"landmark_index\" : index,\n        \"x\" : x,\n        \"y\" : y,\n        \"z\" : z\n    })","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:14:39.740625Z","iopub.execute_input":"2025-06-14T17:14:39.740946Z","iopub.status.idle":"2025-06-14T17:14:39.756458Z","shell.execute_reply.started":"2025-06-14T17:14:39.740920Z","shell.execute_reply":"2025-06-14T17:14:39.755608Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nmetadata = {}\nwith open('/kaggle/input/wlasl-processed/WLASL_v0.3.json' , 'r') as file:\n    metadata = json.load(file)","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:14:44.810682Z","iopub.execute_input":"2025-06-14T17:14:44.811473Z","iopub.status.idle":"2025-06-14T17:14:44.916681Z","shell.execute_reply.started":"2025-06-14T17:14:44.811430Z","shell.execute_reply":"2025-06-14T17:14:44.915983Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-14T11:20:37.163052Z","iopub.execute_input":"2025-06-14T11:20:37.163508Z","iopub.status.idle":"2025-06-14T11:20:40.240889Z","shell.execute_reply.started":"2025-06-14T11:20:37.163476Z","shell.execute_reply":"2025-06-14T11:20:40.239843Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labelMap = {} # new metadata with relevant info\nfor i in metadata:\n    label = i['gloss']\n    for instance in i['instances']:\n        Id = int(instance['video_id'])\n        frame_start = instance['frame_start']\n        frame_end = instance['frame_end']\n        fps = instance['fps']\n        labelMap[Id] = [label , frame_start , frame_end , fps] # this is the info matters to us","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:14:48.227853Z","iopub.execute_input":"2025-06-14T17:14:48.228547Z","iopub.status.idle":"2025-06-14T17:14:48.256770Z","shell.execute_reply.started":"2025-06-14T17:14:48.228498Z","shell.execute_reply":"2025-06-14T17:14:48.255889Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"theoretical_file_count = len(labelMap)\ntheoretical_file_count # words said to be present in the dataset","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:14:54.234867Z","iopub.execute_input":"2025-06-14T17:14:54.235475Z","iopub.status.idle":"2025-06-14T17:14:54.241270Z","shell.execute_reply.started":"2025-06-14T17:14:54.235438Z","shell.execute_reply":"2025-06-14T17:14:54.240442Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labelMap[38540] # sample format of metadata ","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:08:42.521869Z","iopub.execute_input":"2025-06-14T17:08:42.522618Z","iopub.status.idle":"2025-06-14T17:08:42.528021Z","shell.execute_reply.started":"2025-06-14T17:08:42.522584Z","shell.execute_reply":"2025-06-14T17:08:42.527174Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"  # Disable oneDNN optimizations warnings\nimport sys\nimport warnings\nimport logging\nfrom contextlib import redirect_stderr\nimport io\n\n# HARDCORE SUPPRESSION - Put this at the VERY TOP of your script\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nos.environ['TF_ENABLE_ONEDNN_OPTS'] = '0'\nos.environ['CUDA_VISIBLE_DEVICES'] = ''  # Force CPU only to reduce warnings\nos.environ['TF_FORCE_GPU_ALLOW_GROWTH'] = 'true'\n\n# Suppress all warnings\nwarnings.filterwarnings('ignore')\nlogging.getLogger().setLevel(logging.ERROR)\n\n# Redirect stderr to suppress C++ level warnings\nclass SuppressStderr:\n    def __enter__(self):\n        self._original_stderr = sys.stderr\n        sys.stderr = open(os.devnull, 'w')\n        return self\n\n    def __exit__(self, exc_type, exc_val, exc_tb):\n        sys.stderr.close()\n        sys.stderr = self._original_stderr\n\n# Import TensorFlow AFTER setting environment variables\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR')\ntf.compat.v1.logging.set_verbosity(tf.compat.v1.logging.ERROR)\n\n# Suppress ABSL\ntry:\n    import absl.logging\n    absl.logging.set_verbosity(absl.logging.ERROR)\nexcept:\n    pass\n\npath = '/kaggle/working/temp'\n\nvideoPath = []\nvideoId = []\nlabelName = []\n\nif not os.path.exists(path):\n    os.makedirs(path)\n\nvideo_path = '/kaggle/input/wlasl-processed/videos'\nfor video in os.listdir(video_path):\n    if(video.endswith('.mp4')):\n        video_filename = os.path.basename(video)\n        video_id = int(os.path.splitext(video_filename)[0])\n        \n        start_frame = labelMap[video_id][1]\n        end_frame = labelMap[video_id][2]\n        fps = labelMap[video_id][3]\n        \n        df = transform(video , start_frame , end_frame , fps)\n        \n        dest_path = os.path.join(path, f'{video_id}.parquet')\n        df.to_parquet(dest_path)\n        \n        videoPath.append(dest_path)\n        videoId.append(video_id)\n        labelName.append(labelMap[video_id][0])\n\ndata_frame = pd.DataFrame({\n    'video_id' : videoId,\n    'video_path' : videoPath,\n    'label' : labelName\n})\n\ndata_frame.to_csv('/kaggle/working/temp/summary.csv')","metadata":{"execution":{"iopub.status.busy":"2025-06-14T17:15:07.318248Z","iopub.execute_input":"2025-06-14T17:15:07.318573Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nvideo_path = '/kaggle/working/temp'\nc = 0\nfor video in os.listdir(video_path):\n    if(video.endswith('.csv')):\n        c += 1\nc","metadata":{"execution":{"iopub.status.busy":"2023-11-11T10:56:12.402966Z","iopub.execute_input":"2023-11-11T10:56:12.403403Z","iopub.status.idle":"2023-11-11T10:56:12.420957Z","shell.execute_reply.started":"2023-11-11T10:56:12.40336Z","shell.execute_reply":"2023-11-11T10:56:12.420009Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}