{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"},{"sourceId":5600436,"sourceType":"datasetVersion","datasetId":3221731},{"sourceId":8843739,"sourceType":"datasetVersion","datasetId":5322814}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-02T19:21:18.663799Z","iopub.execute_input":"2024-07-02T19:21:18.664191Z","iopub.status.idle":"2024-07-02T19:21:18.670039Z","shell.execute_reply.started":"2024-07-02T19:21:18.664156Z","shell.execute_reply":"2024-07-02T19:21:18.668336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mediapipe==0.9.1.0","metadata":{"execution":{"iopub.status.busy":"2024-07-02T21:29:57.982972Z","iopub.execute_input":"2024-07-02T21:29:57.983362Z","iopub.status.idle":"2024-07-02T21:30:18.258195Z","shell.execute_reply.started":"2024-07-02T21:29:57.983329Z","shell.execute_reply":"2024-07-02T21:30:18.257002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow==2.11.0","metadata":{"execution":{"iopub.status.busy":"2024-07-02T21:37:40.705349Z","iopub.execute_input":"2024-07-02T21:37:40.705833Z","iopub.status.idle":"2024-07-02T21:39:23.036338Z","shell.execute_reply.started":"2024-07-02T21:37:40.705792Z","shell.execute_reply":"2024-07-02T21:39:23.030603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport mediapipe as mp\nimport pandas as pd\nimport uuid\nimport numpy as np\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-07-02T21:39:23.041998Z","iopub.execute_input":"2024-07-02T21:39:23.043163Z","iopub.status.idle":"2024-07-02T21:39:31.651485Z","shell.execute_reply.started":"2024-07-02T21:39:23.043015Z","shell.execute_reply":"2024-07-02T21:39:31.650341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"youtube_sign = 'airplane'\nyoutube_sign_video_path = '/kaggle/input/youtube-signs/plane-AIRPLANE.mp4'","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:04:02.946323Z","iopub.execute_input":"2024-07-02T22:04:02.946791Z","iopub.status.idle":"2024-07-02T22:04:02.953259Z","shell.execute_reply.started":"2024-07-02T22:04:02.946754Z","shell.execute_reply":"2024-07-02T22:04:02.951801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Initialize MediaPipe solutions\nmp_face_mesh = mp.solutions.face_mesh\nmp_hands = mp.solutions.hands\nmp_pose = mp.solutions.pose\n\nface_mesh = mp_face_mesh.FaceMesh(static_image_mode=False, max_num_faces=1, min_detection_confidence=0.5)\nhands = mp_hands.Hands(static_image_mode=False, max_num_hands=2, min_detection_confidence=0.5, min_tracking_confidence=0.5)\npose = mp_pose.Pose(static_image_mode=False, min_detection_confidence=0.5, min_tracking_confidence=0.5)\n\n# Define the video file path\nvideo_path = youtube_sign_video_path\n\n# Print the video file path for debugging\nprint(f\"Video file path: {video_path}\")\n\n# Check if the video file exists\nif not os.path.isfile(video_path):\n    print(\"Error: Video file does not exist.\")\nelse:\n    print(\"Video file exists.\")\n\n# Open the video file\ncap = cv2.VideoCapture(video_path)\n\nif not cap.isOpened():\n    print(\"Error: Could not open video file.\")\nelse:\n    print(\"Video file opened successfully.\")\n\n    data = []\n    frame_count = 0\n\n    # Define the number of landmarks for each type\n    num_face_landmarks = 468\n    num_pose_landmarks = 33\n    num_hand_landmarks = 21\n\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            print(\"End of video file reached or can't read the frame.\")\n            break\n\n        frame_count += 1\n        print(f\"Processing frame {frame_count}\")\n\n        # Convert the frame to RGB\n        rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n\n        # Process the frame to extract face mesh landmarks\n        face_results = face_mesh.process(rgb_frame)\n\n        # Process the frame to extract hand landmarks\n        hand_results = hands.process(rgb_frame)\n\n        # Process the frame to extract pose landmarks\n        pose_results = pose.process(rgb_frame)\n\n        # Initialize landmarks with NaN values\n        for landmark_type, num_landmarks in [\n            ('face', num_face_landmarks),\n            ('left_hand', num_hand_landmarks),\n            ('pose', num_pose_landmarks),\n            ('right_hand', num_hand_landmarks)\n        ]:\n            for idx in range(num_landmarks):\n                row_id = f\"{frame_count}-{landmark_type}-{idx}\"\n                data.append({\n                    'frame': frame_count,\n                    'row_id': row_id,\n                    'type': landmark_type,\n                    'landmark_index': idx,\n                    'x': np.nan,\n                    'y': np.nan,\n                    'z': np.nan\n                })\n\n        # Update landmarks with detected values\n        def update_landmarks(results, landmark_type):\n            if results:\n                for landmark_set in results:\n                    for idx, landmark in enumerate(landmark_set.landmark):\n                        row_id = f\"{frame_count}-{landmark_type}-{idx}\"\n                        for record in data:\n                            if record['row_id'] == row_id:\n                                record['x'] = landmark.x\n                                record['y'] = landmark.y\n                                record['z'] = landmark.z\n\n        # Update face landmarks\n        if face_results.multi_face_landmarks:\n            update_landmarks(face_results.multi_face_landmarks, 'face')\n\n        # Update hand landmarks\n        if hand_results.multi_hand_landmarks:\n            for hand_landmarks, handedness in zip(hand_results.multi_hand_landmarks, hand_results.multi_handedness):\n                hand_type = 'left_hand' if handedness.classification[0].label == 'Left' else 'right_hand'\n                update_landmarks([hand_landmarks], hand_type)\n\n        # Update pose landmarks\n        if pose_results.pose_landmarks:\n            update_landmarks([pose_results.pose_landmarks], 'pose')\n\n        # Debugging: Print collected data for this frame\n        print(f\"Collected {len(data)} landmarks so far\")\n\n# Release resources\ncap.release()\nface_mesh.close()\nhands.close()\npose.close()\n\n# Convert the collected data to a pandas DataFrame\ndf = pd.DataFrame(data)\n\n# Debugging: Print the DataFrame before saving\nprint(df.head())\n\nsaved_file_name = youtube_sign + '.parquet'\n# Save the DataFrame to a parquet file\n# df.to_parquet('landmarks.parquet')\ndf.to_parquet(saved_file_name)\n\nprint(\"Landmarks have been saved to {}\".format(saved_file_name))\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:05:34.421984Z","iopub.execute_input":"2024-07-02T22:05:34.422453Z","iopub.status.idle":"2024-07-02T22:07:51.797260Z","shell.execute_reply.started":"2024-07-02T22:05:34.422400Z","shell.execute_reply":"2024-07-02T22:07:51.795849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\nyoutube_sign_parquet_file = '/kaggle/working/' + saved_file_name\n# Load the parquet file\ndata = pd.read_parquet(youtube_sign_parquet_file)\n\n# Print the first few rows to inspect the data\nprint(data.head())\n\n# Check the unique values in 'type' column to ensure they are as expected\nprint(\"Unique types in 'type' column:\", data['type'].unique())\n\n# Define the frame_id and landmark_type you want to group by\nframe_id = 1\nlandmark_type = 'left_hand'\n\n# Check if the specific combination exists\nprint(data[(data['frame'] == frame_id) & (data['type'] == landmark_type)])\n\n# Group by 'frame' and 'type' and then get the group\ntry:\n    df = data.groupby(['frame', 'type']).get_group((frame_id, landmark_type))\n    print(df)\nexcept KeyError as e:\n    print(f\"KeyError: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:04.971047Z","iopub.execute_input":"2024-07-02T22:08:04.971480Z","iopub.status.idle":"2024-07-02T22:08:05.051186Z","shell.execute_reply.started":"2024-07-02T22:08:04.971447Z","shell.execute_reply":"2024-07-02T22:08:05.049776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib import animation\nfrom pathlib import Path\nimport IPython\nfrom IPython.display import display\nfrom IPython.display import HTML\nimport matplotlib as mpl\n\nfrom mediapipe.framework.formats import landmark_pb2","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:18.170218Z","iopub.execute_input":"2024-07-02T22:08:18.170625Z","iopub.status.idle":"2024-07-02T22:08:18.177882Z","shell.execute_reply.started":"2024-07-02T22:08:18.170592Z","shell.execute_reply":"2024-07-02T22:08:18.176338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Cfg:\n    RANDOM_STATE = 2023\n    INPUT_ROOT = Path('/kaggle/input/asl-signs/')\n    OUTPUT_ROOT = Path('kaggle/working')\n    INDEX_MAP_FILE = INPUT_ROOT / 'sign_to_prediction_index_map.json'\n    TRAN_FILE = INPUT_ROOT / 'train.csv'\n    INDEX = 'sequence_id'\n    ROW_ID = 'row_id'","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:20.142773Z","iopub.execute_input":"2024-07-02T22:08:20.144064Z","iopub.status.idle":"2024-07-02T22:08:20.150150Z","shell.execute_reply.started":"2024-07-02T22:08:20.144021Z","shell.execute_reply":"2024-07-02T22:08:20.148944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'cv2 version: {cv2.__version__}')\nprint(f'MediaPipe version: {mp.__version__}')\nprint(f'IPython version: {IPython.__version__}')","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:22.616890Z","iopub.execute_input":"2024-07-02T22:08:22.617350Z","iopub.status.idle":"2024-07-02T22:08:22.623388Z","shell.execute_reply.started":"2024-07-02T22:08:22.617316Z","shell.execute_reply":"2024-07-02T22:08:22.622314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_index_map(file_path=Cfg.INDEX_MAP_FILE):\n    \"\"\"Reads the sign to predict as json file.\"\"\"\n    with open(file_path, \"r\") as f:\n        result = json.load(f)\n    return result    \n\ndef read_train(file_path=Cfg.TRAN_FILE):\n    \"\"\"Reads the train csv as pandas data frame.\"\"\"\n    return pd.read_csv(file_path).set_index(Cfg.INDEX)\n\ndef read_landmark_data_by_path(file_path, input_root=Cfg.INPUT_ROOT):\n    \"\"\"Reads landmak data by the given file path.\"\"\"\n    data = pd.read_parquet(input_root / file_path)\n    return data.set_index(Cfg.ROW_ID)\n\ndef read_landmark_data_by_id(sequence_id, train_data):\n    \"\"\"Reads the landmark data by the given sequence id.\"\"\"\n    file_path = train_data.loc[sequence_id]['path']\n    return read_landmark_data_by_path(file_path)\n\n\n\nmp_drawing = mp.solutions.drawing_utils\nmp_hands = mp.solutions.hands\nmp_face_mesh = mp.solutions.face_mesh\nmp_pose = mp.solutions.pose\n\ndef get_random_sequence_id(train_data):\n    idx = np.random.randint(0, len(train_data))\n    return train_data.index[idx]\n\n# def create_blank_image(height, width):\n#     return np.zeros((height, width, 3), np.uint8)\n\ndef create_blank_image(height, width):\n    return np.ones((height, width, 3), np.uint8)\n\ndef draw_landmarks(\n    data, \n    image, \n    frame_id, \n    landmark_type, \n    connection_type, \n    landmark_color=(255, 0, 0), \n    connection_color=(0, 20, 255), \n    thickness=1, \n    circle_radius=1\n):\n    \"\"\"Draws landmarks\"\"\"\n    df = data.groupby(['frame', 'type']).get_group((frame_id, landmark_type))\n    landmarks = [landmark_pb2.NormalizedLandmark(x=lm.x, y=lm.y, z=lm.z) for idx, lm in df.iterrows()]\n    landmark_list = landmark_pb2.NormalizedLandmarkList(landmark = landmarks)\n\n    mp_drawing.draw_landmarks(\n        image=image,\n        landmark_list=landmark_list, \n        connections=connection_type,\n        landmark_drawing_spec=mp_drawing.DrawingSpec(\n            color=landmark_color, \n            thickness=thickness, \n            circle_radius=circle_radius),\n        connection_drawing_spec=mp_drawing.DrawingSpec(\n            color=connection_color, \n            thickness=thickness, \n            circle_radius=circle_radius))\n    return image\n\ndef draw_left_hand(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='left_hand', \n        connection_type=mp_hands.HAND_CONNECTIONS,\n        landmark_color=(255, 0, 0),\n#         connection_color=(0, 20, 255),\n        connection_color=(0, 255, 0),\n        thickness=3, \n        circle_radius=3)\n\ndef draw_right_hand(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='right_hand', \n        connection_type=mp_hands.HAND_CONNECTIONS,\n        landmark_color=(255, 0, 0),\n#         connection_color=(0, 20, 255),\n        connection_color=(0, 255, 0),\n        thickness=3, \n        circle_radius=3)\n\ndef draw_face(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='face', \n        connection_type=mp_face_mesh.FACEMESH_TESSELATION,\n        landmark_color=(255, 255, 255),\n        connection_color=(0, 255, 0))      \n    \ndef draw_pose(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='pose', \n        connection_type=mp_pose.POSE_CONNECTIONS,\n#         landmark_color=(255, 255, 255),\n#         connection_color=(255, 0, 0),\n        landmark_color=(255, 0, 0),\n        connection_color=(0, 255, 0),\n        thickness=2, \n        circle_radius=2)\n\ndef create_frame(data, frame_id, height=1000, width=1000):\n    image = create_blank_image(height, width)    \n\n    draw_pose(data, image, frame_id) \n    draw_left_hand(data, image, frame_id)    \n    draw_right_hand(data, image, frame_id)  \n    draw_face(data, image, frame_id)\n     \n    return image","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:25.039984Z","iopub.execute_input":"2024-07-02T22:08:25.040417Z","iopub.status.idle":"2024-07-02T22:08:25.066573Z","shell.execute_reply.started":"2024-07-02T22:08:25.040386Z","shell.execute_reply":"2024-07-02T22:08:25.064995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_frames_any_parquet(filepath, height=800, width=800):\n    data = pd.read_parquet(filepath)\n#     data.set_index(Cfg.ROW_ID)\n    frame_ids = data['frame'].unique()\n    images = [create_frame(data, frame_id=fid, height=height, width=width) for fid in frame_ids]\n    return np.array(images)\n\ndef create_frames(sequence_id, train_data, height=800, width=800):\n    data = read_landmark_data_by_id(sequence_id, train_data)\n    frame_ids = data['frame'].unique()\n    images = [create_frame(data, frame_id=fid, height=height, width=width) for fid in frame_ids]\n    return np.array(images)\n\n\ndef create_animation(images, fig, ax):\n    ax.axis('off')\n    \n    ims = []\n    for img in images:\n        im = ax.imshow(img, animated=True)\n        ims.append([im])\n    \n    func_animation = mpl.animation.ArtistAnimation(\n        fig, \n        ims, \n        interval=100, \n        blit=True,\n        repeat_delay=1000)\n\n    return func_animation\n\ndef get_sign_by_id(sequence_id, train_data):\n    return train_data.loc[sequence_id]['sign']\n\ndef play_animation(sequence_id, train_data, height, width, figsize=(4, 4)):\n    frames = create_frames(sequence_id, train_data, height=height, width=width)\n    sign = get_sign_by_id(sequence_id, train_data)\n    \n    fig, ax = plt.subplots(1, 1, figsize=figsize)\n    anim = create_animation(frames, fig, ax)\n    ax.set_title(f'Sign: {sign}')\n    \n    video = anim.to_html5_video()\n    html = IPython.display.HTML(video)\n    IPython.display.display(html)\n    plt.close()\n\n    \ndef play_animation_any_parquet(filepath,sign_str, height, width, figsize=(4, 4)):\n    frames = create_frames_any_parquet(filepath, height=height, width=width)\n    sign = sign_str\n#     sign = get_sign_by_id(sequence_id, train_data)\n    \n    fig, ax = plt.subplots(1, 1, figsize=figsize)\n    anim = create_animation(frames, fig, ax)\n    ax.set_title(f'Sign: {sign}')\n    \n    video = anim.to_html5_video()\n    html = IPython.display.HTML(video)\n    IPython.display.display(html)\n    plt.close()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:30.647529Z","iopub.execute_input":"2024-07-02T22:08:30.648011Z","iopub.status.idle":"2024-07-02T22:08:30.667784Z","shell.execute_reply.started":"2024-07-02T22:08:30.647974Z","shell.execute_reply":"2024-07-02T22:08:30.666344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"height = 800\nwidth = 600\n\nyoutube_sign_parquet_file = '/kaggle/working/' + saved_file_name\nprint(youtube_sign)\n\nplay_animation_any_parquet(youtube_sign_parquet_file, youtube_sign, height=height, width=width)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:08:33.962652Z","iopub.execute_input":"2024-07-02T22:08:33.963117Z","iopub.status.idle":"2024-07-02T22:08:53.075370Z","shell.execute_reply.started":"2024-07-02T22:08:33.963082Z","shell.execute_reply":"2024-07-02T22:08:53.073977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = read_train()\nheight = 800\nwidth = 600\n\nsequence_id = get_random_sequence_id(train_data)\nplay_animation(sequence_id, train_data, height=height, width=width)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:05:28.760427Z","iopub.status.idle":"2024-07-02T22:05:28.760866Z","shell.execute_reply.started":"2024-07-02T22:05:28.760655Z","shell.execute_reply":"2024-07-02T22:05:28.760673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_gif(sequence_id, train_data, height, width, figsize=(4, 4)):\n    frames = create_frames(sequence_id, train_data, height=height, width=width)\n    sign = get_sign_by_id(sequence_id, train_data)\n    \n    fig, ax = plt.subplots(1, 1, figsize=figsize)\n    anim = create_animation(frames, fig, ax)\n    # Save the animation as a GIF\n    anim.save('sign.gif', writer='imagemagick')\n    plt.close()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:05:28.762467Z","iopub.status.idle":"2024-07-02T22:05:28.762955Z","shell.execute_reply.started":"2024-07-02T22:05:28.762701Z","shell.execute_reply":"2024-07-02T22:05:28.762725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train_data[train_data['sign'] == 'happy']\nsequence_id = get_random_sequence_id(data)\n\ncreate_gif(sequence_id, train_data, height=height, width=width)\n# Display the GIF\nIPython.display.Image(filename='/kaggle/working/sign.gif')","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:05:28.764440Z","iopub.status.idle":"2024-07-02T22:05:28.764844Z","shell.execute_reply.started":"2024-07-02T22:05:28.764642Z","shell.execute_reply":"2024-07-02T22:05:28.764658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 2: use the model to predict the youtube sign video\n\n1. take the content from 1st place reference notebook\n2. do prediction","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport json\nimport os\nfrom multiprocessing import cpu_count\n\ndef read_json_file(file_path):\n    \"\"\"Read a JSON file and parse it into a Python object.\n\n    Args:\n        file_path (str): The path to the JSON file to read.\n\n    Returns:\n        dict: A dictionary object representing the JSON data.\n        \n    Raises:\n        FileNotFoundError: If the specified file path does not exist.\n        ValueError: If the specified file path does not contain valid JSON data.\n    \"\"\"\n    try:\n        # Open the file and load the JSON data into a Python object\n        with open(file_path, 'r') as file:\n            json_data = json.load(file)\n        return json_data\n    except FileNotFoundError:\n        # Raise an error if the file path does not exist\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n    except ValueError:\n        # Raise an error if the file does not contain valid JSON data\n        raise ValueError(f\"Invalid JSON data in file: {file_path}\")\n\ncpu_count()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:08.479046Z","iopub.execute_input":"2024-07-02T22:09:08.479519Z","iopub.status.idle":"2024-07-02T22:09:08.493258Z","shell.execute_reply.started":"2024-07-02T22:09:08.479487Z","shell.execute_reply":"2024-07-02T22:09:08.491742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tf.__version__)\nprint(tf.keras.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:12.143595Z","iopub.execute_input":"2024-07-02T22:09:12.144080Z","iopub.status.idle":"2024-07-02T22:09:12.151360Z","shell.execute_reply.started":"2024-07-02T22:09:12.144045Z","shell.execute_reply":"2024-07-02T22:09:12.149766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\nMAX_LEN = 384\nCROP_LEN = MAX_LEN\nNUM_CLASSES  = 250\nPAD = -100.\nNOSE=[\n    1,2,98,327\n]\nLNOSE = [98]\nRNOSE = [327]\nLIP = [ 0, \n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\nLLIP = [84,181,91,146,61,185,40,39,37,87,178,88,95,78,191,80,81,82]\nRLIP = [314,405,321,375,291,409,270,269,267,317,402,318,324,308,415,310,311,312]\n\nPOSE = [500, 502, 504, 501, 503, 505, 512, 513]\nLPOSE = [513,505,503,501]\nRPOSE = [512,504,502,500]\n\nREYE = [\n    33, 7, 163, 144, 145, 153, 154, 155, 133,\n    246, 161, 160, 159, 158, 157, 173,\n]\nLEYE = [\n    263, 249, 390, 373, 374, 380, 381, 382, 362,\n    466, 388, 387, 386, 385, 384, 398,\n]\n\nLHAND = np.arange(468, 489).tolist()\nRHAND = np.arange(522, 543).tolist()\n\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE #+POSE\n\nNUM_NODES = len(POINT_LANDMARKS)\nCHANNELS = 6*NUM_NODES\n\nprint(NUM_NODES)\nprint(CHANNELS)\n\ndef tf_nan_mean(x, axis=0, keepdims=False):\n    return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis, keepdims=keepdims) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis, keepdims=keepdims)\n\ndef tf_nan_std(x, center=None, axis=0, keepdims=False):\n    if center is None:\n        center = tf_nan_mean(x, axis=axis,  keepdims=True)\n    d = x - center\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis, keepdims=keepdims))\n\nclass Preprocess(tf.keras.layers.Layer):\n    def __init__(self, max_len=MAX_LEN, point_landmarks=POINT_LANDMARKS, **kwargs):\n        super().__init__(**kwargs)\n        self.max_len = max_len\n        self.point_landmarks = point_landmarks\n\n    def call(self, inputs):\n        if tf.rank(inputs) == 3:\n            x = inputs[None,...]\n        else:\n            x = inputs\n        \n        mean = tf_nan_mean(tf.gather(x, [17], axis=2), axis=[1,2], keepdims=True)\n        mean = tf.where(tf.math.is_nan(mean), tf.constant(0.5,x.dtype), mean)\n        x = tf.gather(x, self.point_landmarks, axis=2) #N,T,P,C\n        std = tf_nan_std(x, center=mean, axis=[1,2], keepdims=True)\n        \n        x = (x - mean)/std\n\n        if self.max_len is not None:\n            x = x[:,:self.max_len]\n        length = tf.shape(x)[1]\n        x = x[...,:2]\n\n        dx = tf.cond(tf.shape(x)[1]>1,lambda:tf.pad(x[:,1:] - x[:,:-1], [[0,0],[0,1],[0,0],[0,0]]),lambda:tf.zeros_like(x))\n\n        dx2 = tf.cond(tf.shape(x)[1]>2,lambda:tf.pad(x[:,2:] - x[:,:-2], [[0,0],[0,2],[0,0],[0,0]]),lambda:tf.zeros_like(x))\n\n        x = tf.concat([\n            tf.reshape(x, (-1,length,2*len(self.point_landmarks))),\n            tf.reshape(dx, (-1,length,2*len(self.point_landmarks))),\n            tf.reshape(dx2, (-1,length,2*len(self.point_landmarks))),\n        ], axis = -1)\n        \n        x = tf.where(tf.math.is_nan(x),tf.constant(0.,x.dtype),x)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:15.658472Z","iopub.execute_input":"2024-07-02T22:09:15.659012Z","iopub.status.idle":"2024-07-02T22:09:15.696078Z","shell.execute_reply.started":"2024-07-02T22:09:15.658975Z","shell.execute_reply":"2024-07-02T22:09:15.694618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ECA(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=5, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.kernel_size = kernel_size\n        self.conv = tf.keras.layers.Conv1D(1, kernel_size=kernel_size, strides=1, padding=\"same\", use_bias=False)\n\n    def call(self, inputs, mask=None):\n        nn = tf.keras.layers.GlobalAveragePooling1D()(inputs, mask=mask)\n        nn = tf.expand_dims(nn, -1)\n        nn = self.conv(nn)\n        nn = tf.squeeze(nn, -1)\n        nn = tf.nn.sigmoid(nn)\n        nn = nn[:,None,:]\n        return inputs * nn\n\nclass LateDropout(tf.keras.layers.Layer):\n    def __init__(self, rate, noise_shape=None, start_step=0, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.rate = rate\n        self.start_step = start_step\n        self.dropout = tf.keras.layers.Dropout(rate, noise_shape=noise_shape)\n      \n    def build(self, input_shape):\n        super().build(input_shape)\n        agg = tf.VariableAggregation.ONLY_FIRST_REPLICA\n        self._train_counter = tf.Variable(0, dtype=\"int64\", aggregation=agg, trainable=False)\n\n    def call(self, inputs, training=False):\n        x = tf.cond(self._train_counter < self.start_step, lambda:inputs, lambda:self.dropout(inputs, training=training))\n        if training:\n            self._train_counter.assign_add(1)\n        return x\n\nclass CausalDWConv1D(tf.keras.layers.Layer):\n    def __init__(self, \n        kernel_size=17,\n        dilation_rate=1,\n        use_bias=False,\n        depthwise_initializer='glorot_uniform',\n        name='', **kwargs):\n        super().__init__(name=name,**kwargs)\n        self.causal_pad = tf.keras.layers.ZeroPadding1D((dilation_rate*(kernel_size-1),0),name=name + '_pad')\n        self.dw_conv = tf.keras.layers.DepthwiseConv1D(\n                            kernel_size,\n                            strides=1,\n                            dilation_rate=dilation_rate,\n                            padding='valid',\n                            use_bias=use_bias,\n                            depthwise_initializer=depthwise_initializer,\n                            name=name + '_dwconv')\n        self.supports_masking = True\n        \n    def call(self, inputs):\n        x = self.causal_pad(inputs)\n        x = self.dw_conv(x)\n        return x\n\ndef Conv1DBlock(channel_size,\n          kernel_size,\n          dilation_rate=1,\n          drop_rate=0.0,\n          expand_ratio=2,\n          se_ratio=0.25,\n          activation='swish',\n          name=None):\n    '''\n    efficient conv1d block, @hoyso48\n    '''\n    if name is None:\n        name = str(tf.keras.backend.get_uid(\"mbblock\"))\n    # Expansion phase\n    def apply(inputs):\n        channels_in = tf.keras.backend.int_shape(inputs)[-1]\n        channels_expand = channels_in * expand_ratio\n\n        skip = inputs\n\n        x = tf.keras.layers.Dense(\n            channels_expand,\n            use_bias=True,\n            activation=activation,\n            name=name + '_expand_conv')(inputs)\n\n        # Depthwise Convolution\n        x = CausalDWConv1D(kernel_size,\n            dilation_rate=dilation_rate,\n            use_bias=False,\n            name=name + '_dwconv')(x)\n\n        x = tf.keras.layers.BatchNormalization(momentum=0.95, name=name + '_bn')(x)\n\n        x  = ECA()(x)\n\n        x = tf.keras.layers.Dense(\n            channel_size,\n            use_bias=True,\n            name=name + '_project_conv')(x)\n\n        if drop_rate > 0:\n            x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1), name=name + '_drop')(x)\n\n        if (channels_in == channel_size):\n            x = tf.keras.layers.add([x, skip], name=name + '_add')\n        return x\n\n    return apply","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:19.673339Z","iopub.execute_input":"2024-07-02T22:09:19.674875Z","iopub.status.idle":"2024-07-02T22:09:19.704458Z","shell.execute_reply.started":"2024-07-02T22:09:19.674823Z","shell.execute_reply":"2024-07-02T22:09:19.703056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MultiHeadSelfAttention(tf.keras.layers.Layer):\n    def __init__(self, dim=256, num_heads=4, dropout=0, **kwargs):\n        super().__init__(**kwargs)\n        self.dim = dim\n        self.scale = self.dim ** -0.5\n        self.num_heads = num_heads\n        self.qkv = tf.keras.layers.Dense(3 * dim, use_bias=False)\n        self.drop1 = tf.keras.layers.Dropout(dropout)\n        self.proj = tf.keras.layers.Dense(dim, use_bias=False)\n        self.supports_masking = True\n\n    def call(self, inputs, mask=None):\n        qkv = self.qkv(inputs)\n        qkv = tf.keras.layers.Permute((2, 1, 3))(tf.keras.layers.Reshape((-1, self.num_heads, self.dim * 3 // self.num_heads))(qkv))\n        q, k, v = tf.split(qkv, [self.dim // self.num_heads] * 3, axis=-1)\n\n        attn = tf.matmul(q, k, transpose_b=True) * self.scale\n\n        if mask is not None:\n            mask = mask[:, None, None, :]\n\n        attn = tf.keras.layers.Softmax(axis=-1)(attn, mask=mask)\n        attn = self.drop1(attn)\n\n        x = attn @ v\n        x = tf.keras.layers.Reshape((-1, self.dim))(tf.keras.layers.Permute((2, 1, 3))(x))\n        x = self.proj(x)\n        return x\n\n\ndef TransformerBlock(dim=256, num_heads=4, expand=4, attn_dropout=0.2, drop_rate=0.2, activation='swish'):\n    def apply(inputs):\n        x = inputs\n        x = tf.keras.layers.BatchNormalization(momentum=0.95)(x)\n        x = MultiHeadSelfAttention(dim=dim,num_heads=num_heads,dropout=attn_dropout)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1))(x)\n        x = tf.keras.layers.Add()([inputs, x])\n        attn_out = x\n\n        x = tf.keras.layers.BatchNormalization(momentum=0.95)(x)\n        x = tf.keras.layers.Dense(dim*expand, use_bias=False, activation=activation)(x)\n        x = tf.keras.layers.Dense(dim, use_bias=False)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1))(x)\n        x = tf.keras.layers.Add()([attn_out, x])\n        return x\n    return apply","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:23.384137Z","iopub.execute_input":"2024-07-02T22:09:23.385295Z","iopub.status.idle":"2024-07-02T22:09:23.406628Z","shell.execute_reply.started":"2024-07-02T22:09:23.385248Z","shell.execute_reply":"2024-07-02T22:09:23.405006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model(max_len=MAX_LEN, dropout_step=0, dim=192):\n    inp = tf.keras.Input((max_len,CHANNELS))\n#     x = tf.keras.layers.Masking(mask_value=PAD,input_shape=(max_len,CHANNELS))(inp) #we don't need masking layer with inference\n    x = inp\n    ksize = 17\n    x = tf.keras.layers.Dense(dim, use_bias=False,name='stem_conv')(x)\n    x = tf.keras.layers.BatchNormalization(momentum=0.95,name='stem_bn')(x)\n\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = TransformerBlock(dim,expand=2)(x)\n\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n    x = TransformerBlock(dim,expand=2)(x)\n\n    if dim == 384: #for the 4x sized model\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = TransformerBlock(dim,expand=2)(x)\n\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = Conv1DBlock(dim,ksize,drop_rate=0.2)(x)\n        x = TransformerBlock(dim,expand=2)(x)\n\n    x = tf.keras.layers.Dense(dim*2,activation=None,name='top_conv')(x)\n    x = tf.keras.layers.GlobalAveragePooling1D()(x)\n    x = LateDropout(0.8, start_step=dropout_step)(x)\n    x = tf.keras.layers.Dense(NUM_CLASSES,name='classifier')(x)\n    return tf.keras.Model(inp, x)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:26.745310Z","iopub.execute_input":"2024-07-02T22:09:26.745747Z","iopub.status.idle":"2024-07-02T22:09:26.763151Z","shell.execute_reply.started":"2024-07-02T22:09:26.745714Z","shell.execute_reply":"2024-07-02T22:09:26.761534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use original author's one model for tensorflow js model\nmodels_path = [\n              '/kaggle/input/islr-models/islr-fp16-192-8-seed42-foldall-last.h5', #comment out other weights to check single model score\n               '/kaggle/input/islr-models/islr-fp16-192-8-seed43-foldall-last.h5',\n               '/kaggle/input/islr-models/islr-fp16-192-8-seed44-foldall-last.h5',\n               '/kaggle/input/islr-models/islr-fp16-192-8-seed45-foldall-last.h5',\n              ]\n\n# models_path = ['/kaggle/input/reproduce-1stplace-islr-google-foldall-seed44/hm-islr-fp16-192-8-seed44-foldall-last.h5']\n\nmodels = [get_model() for _ in models_path]\nfor model,path in zip(models,models_path):\n    model.load_weights(path)\nmodels[0].summary()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:30.317421Z","iopub.execute_input":"2024-07-02T22:09:30.317836Z","iopub.status.idle":"2024-07-02T22:09:36.837839Z","shell.execute_reply.started":"2024-07-02T22:09:30.317803Z","shell.execute_reply":"2024-07-02T22:09:36.836600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TFLiteModel(tf.Module):\n    \"\"\"\n    TensorFlow Lite model that takes input tensors and applies:\n        – a preprocessing model\n        – the ISLR model \n    \"\"\"\n\n    def __init__(self, islr_models):\n        \"\"\"\n        Initializes the TFLiteModel with the specified preprocessing model and ISLR model.\n        \"\"\"\n        super(TFLiteModel, self).__init__()\n\n        # Load the feature generation and main models\n        self.prep_inputs = Preprocess()\n        self.islr_models   = islr_models\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs):\n        \"\"\"\n        Applies the feature generation model and main model to the input tensors.\n\n        Args:\n            inputs: Input tensor with shape [batch_size, 543, 3].\n\n        Returns:\n            A dictionary with a single key 'outputs' and corresponding output tensor.\n        \"\"\"\n        x = self.prep_inputs(tf.cast(inputs, dtype=tf.float32))\n        outputs = [model(x) for model in self.islr_models]\n        outputs = tf.keras.layers.Average()(outputs)[0]\n        return {'outputs': outputs}","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:42.400824Z","iopub.execute_input":"2024-07-02T22:09:42.401298Z","iopub.status.idle":"2024-07-02T22:09:42.413001Z","shell.execute_reply.started":"2024-07-02T22:09:42.401262Z","shell.execute_reply":"2024-07-02T22:09:42.411727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet('/kaggle/input/asl-signs/' + pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:45.436551Z","iopub.execute_input":"2024-07-02T22:09:45.437127Z","iopub.status.idle":"2024-07-02T22:09:45.444972Z","shell.execute_reply.started":"2024-07-02T22:09:45.437081Z","shell.execute_reply":"2024-07-02T22:09:45.443439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.path.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:05:28.788310Z","iopub.status.idle":"2024-07-02T22:05:28.788718Z","shell.execute_reply.started":"2024-07-02T22:05:28.788519Z","shell.execute_reply":"2024-07-02T22:05:28.788537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/asl-signs/train.csv')\nprint(\"\\n\\n... LOAD SIGN TO PREDICTION INDEX MAP FROM JSON FILE ...\\n\")\ns2p_map = {k.lower():v for k,v in read_json_file(os.path.join(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\")).items()}\np2s_map = {v:k for k,v in read_json_file(os.path.join(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\")).items()}\nencoder = lambda x: s2p_map.get(x.lower())\ndecoder = lambda x: p2s_map.get(x)\n# print(s2p_map)\ntrain_df['label'] = train_df.sign.map(encoder)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:49.098165Z","iopub.execute_input":"2024-07-02T22:09:49.099518Z","iopub.status.idle":"2024-07-02T22:09:49.375526Z","shell.execute_reply.started":"2024-07-02T22:09:49.099476Z","shell.execute_reply":"2024-07-02T22:09:49.374111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict correctly first example of the training set\ntflite_keras_model = TFLiteModel(islr_models=models)\ndemo_output = tflite_keras_model(load_relevant_data_subset(train_df.path[0]))[\"outputs\"]\ndecoder(np.argmax(demo_output.numpy(), axis=-1))","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:09:53.579977Z","iopub.execute_input":"2024-07-02T22:09:53.580418Z","iopub.status.idle":"2024-07-02T22:09:59.525218Z","shell.execute_reply.started":"2024-07-02T22:09:53.580384Z","shell.execute_reply":"2024-07-02T22:09:59.523979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\ndef load_js_cam_parquet(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:10:02.019062Z","iopub.execute_input":"2024-07-02T22:10:02.019486Z","iopub.status.idle":"2024-07-02T22:10:02.027142Z","shell.execute_reply.started":"2024-07-02T22:10:02.019454Z","shell.execute_reply":"2024-07-02T22:10:02.025696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict wrong for the youtube thank you sign.\n# you can try another youtube video as well.\nyoutube_sign = 'airplane'\nyoutube_sign_parquet_file = '/kaggle/working/' + youtube_sign + '.parquet'\nyoutube_sign_output = tflite_keras_model(load_js_cam_parquet(youtube_sign_parquet_file))[\"outputs\"]\ndecoder(np.argmax(youtube_sign_output.numpy(), axis=-1))","metadata":{"execution":{"iopub.status.busy":"2024-07-02T22:10:24.307499Z","iopub.execute_input":"2024-07-02T22:10:24.308098Z","iopub.status.idle":"2024-07-02T22:10:24.358660Z","shell.execute_reply.started":"2024-07-02T22:10:24.308052Z","shell.execute_reply":"2024-07-02T22:10:24.357351Z"},"trusted":true},"execution_count":null,"outputs":[]}]}