{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"},{"sourceId":2632847,"sourceType":"datasetVersion","datasetId":1589971},{"sourceId":4747505,"sourceType":"datasetVersion","datasetId":2747345},{"sourceId":5315518,"sourceType":"datasetVersion","datasetId":3036481},{"sourceId":10062629,"sourceType":"datasetVersion","datasetId":6200819},{"sourceId":10092435,"sourceType":"datasetVersion","datasetId":6223508}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hello Fellow Kagglers,\n\nThis notebook demonstrates the data processing and training process in Tensorflow.\n\nI am excited about this competition, because my Master Thesis was on sign language recognition.\n\n**Data Processing**\n\nOnly lips, hands and arm pose coordinates are used.\n\nA custom Tensorflow layer handles the data processing. In short, it filters all frames without coordinates for the hands and downsamples the input to 32 frames if it is too long.\n\n**Model**\n\nA transformer based model is used. The embedding layer makes an ambedding per landmark(lips/left hand/right hand/arm pose) and merges these embedding with fully connected layers. The transformer consists of just 2 blocks with a simple mean pooling and fully connected layers for classification.\n\n\n**V2**\n\n* Learnable attention weights for each landmark\n* Removed layer normalisation in embedding to prevent double layer normalisation at the end of embedding and start of transformer\n* Removed additional fully connected layer in head before classification layer\n\n**V3**\n\n* Using all data for training\n* Increased final embedding size 384 -> 512\n* Added 10% dropout in classification layer\n* Increased number of epoch 50 -> 100\n* Number of transformer heads 8 -> 4\n\n**V5**\n\n* Corrected `USE_VAL` comments thanks to comment from [Jackson You](https://www.kaggle.com/jacksonyou)\n* Corrected bug in preprocessing layer thanks to comment from [bilzard](https://www.kaggle.com/tatamikenn)\n* Increased `INPUT_SIZE` 32 -> 64 and added label smoothing based on this [notebook](https://www.kaggle.com/code/hengck23/lb-0-73-single-fold-transformer-architecture) by [hengck23](https://www.kaggle.com/hengck23)\n* Removed layer normalisation in transformer blocks. Model is very shallow. Conventional transformers have 10s of blocks, GPT-3 for example 96 and ViT-Huge 32, in contrast to this tiny transformer of just 2 blocks. The layer normalisation does not seem to be needed in such shallow networks.\n* Biggest difference, normalisation to a single hand. When normalised to left handed, less than 1 percent of the right handed frames was filled. The model consists of only the lips, left hand and left arm pose.\n\nV5 will most likely be the last update to keep this competition competitive.\n\nGood luck to all of you in the last month of this exciting competition!\n\nIf you have any feedback or questions, please feel free to leave a comment.\n\nExpect updates in the coming weeks!","metadata":{}},{"cell_type":"code","source":"!pip install mediapipe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:23.115561Z","iopub.execute_input":"2024-12-05T23:23:23.115932Z","iopub.status.idle":"2024-12-05T23:23:31.199723Z","shell.execute_reply.started":"2024-12-05T23:23:23.115885Z","shell.execute_reply":"2024-12-05T23:23:31.198856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\n# import tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport scipy\n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:31.201667Z","iopub.execute_input":"2024-12-05T23:23:31.202345Z","iopub.status.idle":"2024-12-05T23:23:34.584439Z","shell.execute_reply.started":"2024-12-05T23:23:31.202299Z","shell.execute_reply":"2024-12-05T23:23:34.583586Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Plot Config","metadata":{}},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.585488Z","iopub.execute_input":"2024-12-05T23:23:34.585962Z","iopub.status.idle":"2024-12-05T23:23:34.591057Z","shell.execute_reply.started":"2024-12-05T23:23:34.585930Z","shell.execute_reply":"2024-12-05T23:23:34.590218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"# If True, processing data from scratch\n# If False, loads preprocessed data\nPREPROCESS_DATA = False\nTRAIN_MODEL = True\n# True: use 10% of participants as validation set\n# False: use all data for training -> gives better LB result\nUSE_VAL = True\n\nN_ROWS = 543\nN_DIMS = 3\nDIM_NAMES = ['x', 'y', 'z']\nSEED = 42\nNUM_CLASSES = 250\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\nINPUT_SIZE = 64\n\nBATCH_ALL_SIGNS_N = 4\nBATCH_SIZE = 256\nN_EPOCHS = 5\nLR_MAX = 1e-3\nN_WARMUP_EPOCHS = 0\nWD_RATIO = 0.05\nMASK_VAL = 4237","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.593595Z","iopub.execute_input":"2024-12-05T23:23:34.593995Z","iopub.status.idle":"2024-12-05T23:23:34.604922Z","shell.execute_reply.started":"2024-12-05T23:23:34.593968Z","shell.execute_reply":"2024-12-05T23:23:34.604242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.605874Z","iopub.execute_input":"2024-12-05T23:23:34.606156Z","iopub.status.idle":"2024-12-05T23:23:34.615027Z","shell.execute_reply.started":"2024-12-05T23:23:34.606131Z","shell.execute_reply":"2024-12-05T23:23:34.614128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"# Read Training Data\n# if IS_INTERACTIVE or not PREPROCESS_DATA:\n#     train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\n# else:\n#     train = pd.read_csv('/kaggle/input/asl-signs/train.csv')\ntrain = pd.read_csv('/kaggle/input/asl-signs/train.csv')\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.616026Z","iopub.execute_input":"2024-12-05T23:23:34.616307Z","iopub.status.idle":"2024-12-05T23:23:34.732487Z","shell.execute_reply.started":"2024-12-05T23:23:34.616281Z","shell.execute_reply":"2024-12-05T23:23:34.731619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Add File Path","metadata":{}},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.733661Z","iopub.execute_input":"2024-12-05T23:23:34.734302Z","iopub.status.idle":"2024-12-05T23:23:34.765322Z","shell.execute_reply.started":"2024-12-05T23:23:34.734256Z","shell.execute_reply":"2024-12-05T23:23:34.764703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ordinally Encode Sign","metadata":{}},{"cell_type":"code","source":"# Add ordinally Encoded Sign (assign number to each sign name)\ntrain['sign_ord'] = train['sign'].astype('category').cat.codes\n\n# Dictionaries to translate sign <-> ordinal encoded sign\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.766421Z","iopub.execute_input":"2024-12-05T23:23:34.766781Z","iopub.status.idle":"2024-12-05T23:23:34.857110Z","shell.execute_reply.started":"2024-12-05T23:23:34.766722Z","shell.execute_reply":"2024-12-05T23:23:34.856468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train.head(30))\ndisplay(train.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.857929Z","iopub.execute_input":"2024-12-05T23:23:34.858154Z","iopub.status.idle":"2024-12-05T23:23:34.891853Z","shell.execute_reply.started":"2024-12-05T23:23:34.858130Z","shell.execute_reply":"2024-12-05T23:23:34.891067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Video Statistics","metadata":{}},{"cell_type":"code","source":"N = int(1e3) if (IS_INTERACTIVE or not PREPROCESS_DATA) else int(10e3)\nN_UNIQUE_FRAMES = np.zeros(N, dtype=np.uint16)\nN_MISSING_FRAMES = np.zeros(N, dtype=np.uint16)\nMAX_FRAME = np.zeros(N, dtype=np.uint16)\n\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999]\n\nfor idx, file_path in enumerate(tqdm(train['file_path'].sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    N_UNIQUE_FRAMES[idx] = df['frame'].nunique()\n    N_MISSING_FRAMES[idx] = (df['frame'].max() - df['frame'].min()) - df['frame'].nunique() + 1\n    MAX_FRAME[idx] = df['frame'].max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:34.895286Z","iopub.execute_input":"2024-12-05T23:23:34.895874Z","iopub.status.idle":"2024-12-05T23:23:44.780851Z","shell.execute_reply.started":"2024-12-05T23:23:34.895846Z","shell.execute_reply":"2024-12-05T23:23:44.779803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_UNIQUE_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+25, 25))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:44.782146Z","iopub.execute_input":"2024-12-05T23:23:44.782437Z","iopub.status.idle":"2024-12-05T23:23:45.212583Z","shell.execute_reply.started":"2024-12-05T23:23:44.782408Z","shell.execute_reply":"2024-12-05T23:23:45.211805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of missing frames, consecutive frames with missing intermediate frame, i.e. 1,2,4,5 -> 3 is missing\ndisplay(pd.Series(N_MISSING_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_MISSING_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Missing Frames', size=24)\npd.Series(N_MISSING_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:45.213800Z","iopub.execute_input":"2024-12-05T23:23:45.214163Z","iopub.status.idle":"2024-12-05T23:23:45.766831Z","shell.execute_reply.started":"2024-12-05T23:23:45.214122Z","shell.execute_reply":"2024-12-05T23:23:45.765956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:45.768066Z","iopub.execute_input":"2024-12-05T23:23:45.768391Z","iopub.status.idle":"2024-12-05T23:23:46.151660Z","shell.execute_reply.started":"2024-12-05T23:23:45.768362Z","shell.execute_reply":"2024-12-05T23:23:46.150905Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Landmark Indices","metadata":{}},{"cell_type":"code","source":"USE_TYPES = ['left_hand', 'pose', 'right_hand']\nSTART_IDX = 468\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n# Landmark indices in original data\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n# Landmark indices in processed data\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\nprint(f'# HAND_IDXS: {len(HAND_IDXS)}, N_COLS: {N_COLS}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.152980Z","iopub.execute_input":"2024-12-05T23:23:46.153343Z","iopub.status.idle":"2024-12-05T23:23:46.162707Z","shell.execute_reply.started":"2024-12-05T23:23:46.153302Z","shell.execute_reply":"2024-12-05T23:23:46.161814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.163906Z","iopub.execute_input":"2024-12-05T23:23:46.164560Z","iopub.status.idle":"2024-12-05T23:23:46.173977Z","shell.execute_reply.started":"2024-12-05T23:23:46.164518Z","shell.execute_reply":"2024-12-05T23:23:46.173127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Process Data Tensorflow","metadata":{}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/competitions/asl-signs/overview/evaluation\nROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.175081Z","iopub.execute_input":"2024-12-05T23:23:46.175349Z","iopub.status.idle":"2024-12-05T23:23:46.182580Z","shell.execute_reply.started":"2024-12-05T23:23:46.175323Z","shell.execute_reply":"2024-12-05T23:23:46.181896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        normalisation_correction = tf.constant([\n                    # Add 0.50 to left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                    [0] * len(LIPS_IDXS) + [0.50] * len(LEFT_HAND_IDXS) + [0.50] * len(POSE_IDXS),\n                    # Y coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                    # Z coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                ],\n                dtype=tf.float32,\n            )\n        self.normalisation_correction = tf.transpose(normalisation_correction, [1,0])\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_ROWS,N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand by comparing summed absolute coordinates\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non NaN Hand values in each frame for the dominant hand\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        \n        # Find frames indices with coordinates of dominant hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter frames\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LANDMARK_IDXS_LEFT_DOMINANT0, axis=1)\n        else:\n            data = tf.gather(data, LANDMARK_IDXS_RIGHT_DOMINANT0, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n        \n        # Video fits in INPUT_SIZE\n        if N_FRAMES < INPUT_SIZE:\n            # Pad With -1 to indicate padding\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            # Pad Data With Zeros\n            data = tf.pad(data, [[0, INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        # Video needs to be downsampled to INPUT_SIZE\n        else:\n            # Repeat\n            if N_FRAMES < INPUT_SIZE**2:\n                repeats = tf.math.floordiv(INPUT_SIZE * INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), INPUT_SIZE)\n            if tf.math.mod(len(data), INPUT_SIZE) > 0:\n                pool_size += 1\n\n            if pool_size == 1:\n                pad_size = (pool_size * INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [INPUT_SIZE, -1, N_COLS, N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.183678Z","iopub.execute_input":"2024-12-05T23:23:46.184022Z","iopub.status.idle":"2024-12-05T23:23:46.738680Z","shell.execute_reply.started":"2024-12-05T23:23:46.183995Z","shell.execute_reply":"2024-12-05T23:23:46.737841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Interpolate NaN Values","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n        \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.740071Z","iopub.execute_input":"2024-12-05T23:23:46.740823Z","iopub.status.idle":"2024-12-05T23:23:46.747539Z","shell.execute_reply.started":"2024-12-05T23:23:46.740778Z","shell.execute_reply":"2024-12-05T23:23:46.746788Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create Dataset","metadata":{}},{"cell_type":"code","source":"# Get the full dataset\ndef preprocess_data():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, INPUT_SIZE], -1, dtype=np.float32)\n\n    # Fill X/y\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        # Log message every 5000 samples\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        # Sanity check, data should not contain NaN values\n        if np.isnan(data).sum() > 0:\n            print(row_idx)\n            return data\n\n    # Save X/y\n    np.save('X.npy', X)\n    np.save('y.npy', y)\n    np.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    \n    # Save Validation\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\n    PARTICIPANT_IDS = train['participant_id'].values\n    train_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n    # Save Train\n    X_train = X[train_idxs]\n    NON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\n    y_train = y[train_idxs]\n    np.save('X_train.npy', X_train)\n    np.save('y_train.npy', y_train)\n    np.save('NON_EMPTY_FRAME_IDXS_TRAIN.npy', NON_EMPTY_FRAME_IDXS_TRAIN)\n    # Save Validation\n    X_val = X[val_idxs]\n    NON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\n    y_val = y[val_idxs]\n    np.save('X_val.npy', X_val)\n    np.save('y_val.npy', y_val)\n    np.save('NON_EMPTY_FRAME_IDXS_VAL.npy', NON_EMPTY_FRAME_IDXS_VAL)\n    # Split Statistics\n    print(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n    print(f'X_train shape: {X_train.shape}, X_val shape: {X_val.shape}')\n    print(f'y_train shape: {y_train.shape}, y_val shape: {y_val.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.748587Z","iopub.execute_input":"2024-12-05T23:23:46.748933Z","iopub.status.idle":"2024-12-05T23:23:46.760260Z","shell.execute_reply.started":"2024-12-05T23:23:46.748888Z","shell.execute_reply":"2024-12-05T23:23:46.759540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess All Data From Scratch\nif PREPROCESS_DATA:\n    preprocess_data()\n    ROOT_DIR = '.'\nelse:\n    ROOT_DIR = '/kaggle/input/gislr-dataset-public'\n    \n# Load Data\nif USE_VAL:\n    # Load Train\n    X_train = np.load(f'{ROOT_DIR}/X_train.npy')\n    y_train = np.load(f'{ROOT_DIR}/y_train.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_TRAIN.npy')\n    # Load Val\n    X_val = np.load(f'{ROOT_DIR}/X_val.npy')\n    y_val = np.load(f'{ROOT_DIR}/y_val.npy')\n    NON_EMPTY_FRAME_IDXS_VAL = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_VAL.npy')\n    # Define validation Data\n    validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val)\nelse:\n    X_train = np.load(f'{ROOT_DIR}/X.npy')\n    y_train = np.load(f'{ROOT_DIR}/y.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS.npy')\n    validation_data = None\n\n# Train \nprint_shape_dtype([X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN], ['X_train', 'y_train', 'NON_EMPTY_FRAME_IDXS_TRAIN'])\n# Val\nif USE_VAL:\n    print_shape_dtype([X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL], ['X_val', 'y_val', 'NON_EMPTY_FRAME_IDXS_VAL'])\n# Sanity Check\nprint(f'# NaN Values X_train: {np.isnan(X_train).sum()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:46.761303Z","iopub.execute_input":"2024-12-05T23:23:46.761673Z","iopub.status.idle":"2024-12-05T23:23:49.685738Z","shell.execute_reply.started":"2024-12-05T23:23:46.761624Z","shell.execute_reply":"2024-12-05T23:23:49.684709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class Count\ndisplay(pd.Series(y_train).value_counts().to_frame('Class Count').iloc[[0,1,2,3,4, -5,-4,-3,-2,-1]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:49.686741Z","iopub.execute_input":"2024-12-05T23:23:49.687047Z","iopub.status.idle":"2024-12-05T23:23:49.696801Z","shell.execute_reply.started":"2024-12-05T23:23:49.687019Z","shell.execute_reply":"2024-12-05T23:23:49.695883Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Number Of Frames","metadata":{}},{"cell_type":"code","source":"# Vast majority of samples fits has less than 32 non empty frames\nN_EMPTY_FRAMES = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum(axis=1) \nN_EMPTY_FRAMES_WATERFALL = []\nfor n in tqdm(range(1,INPUT_SIZE+1)):\n    N_EMPTY_FRAMES_WATERFALL.append(sum(N_EMPTY_FRAMES >= n) / len(NON_EMPTY_FRAME_IDXS_TRAIN) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Non Empty Frames')\npd.Series(N_EMPTY_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks(np.arange(INPUT_SIZE), np.arange(1, INPUT_SIZE+1))\nplt.xlabel('Number of Non Empty Frames', size=16)\nplt.yticks(np.arange(0, 100+10, 10))\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Non Empty Frames', size=16)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:49.698045Z","iopub.execute_input":"2024-12-05T23:23:49.698376Z","iopub.status.idle":"2024-12-05T23:23:50.811642Z","shell.execute_reply.started":"2024-12-05T23:23:49.698338Z","shell.execute_reply":"2024-12-05T23:23:50.810804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Percentage of Frames Filled","metadata":{}},{"cell_type":"code","source":"# Percentage of frames filled, this is the maximum fill percentage of each landmark\nP_DATA_FILLED = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum() / NON_EMPTY_FRAME_IDXS_TRAIN.size * 100\nprint(f'P_DATA_FILLED: {P_DATA_FILLED:.2f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:50.813072Z","iopub.execute_input":"2024-12-05T23:23:50.813741Z","iopub.status.idle":"2024-12-05T23:23:50.826095Z","shell.execute_reply.started":"2024-12-05T23:23:50.813695Z","shell.execute_reply":"2024-12-05T23:23:50.825274Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Statistics - Lips","metadata":{}},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_LEFT_LIPS_MEASUREMENTS = (X_train[:,:,LIPS_IDXS] != 0).sum() / X_train[:,:,LIPS_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_LIPS_MEASUREMENTS: {P_LEFT_LIPS_MEASUREMENTS:.2f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:23:50.827360Z","iopub.execute_input":"2024-12-05T23:23:50.827619Z","iopub.status.idle":"2024-12-05T23:24:03.267670Z","shell.execute_reply.started":"2024-12-05T23:23:50.827592Z","shell.execute_reply":"2024-12-05T23:24:03.266709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_lips_mean_std():\n    # LIPS\n    LIPS_MEAN_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_MEAN_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LIPS_MEAN_X[col] = v.mean()\n                LIPS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LIPS_MEAN_Y[col] = v.mean()\n                LIPS_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Lips {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\n    LIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T\n    \n    return LIPS_MEAN, LIPS_STD\n\nLIPS_MEAN, LIPS_STD = get_lips_mean_std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:03.268854Z","iopub.execute_input":"2024-12-05T23:24:03.269120Z","iopub.status.idle":"2024-12-05T23:24:19.398795Z","shell.execute_reply.started":"2024-12-05T23:24:03.269094Z","shell.execute_reply":"2024-12-05T23:24:19.397911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Statistics - Hands","metadata":{}},{"cell_type":"code","source":"# Verify Normalised to Left Hand Dominant\nP_LEFT_HAND_MEASUREMENTS = (X_train[:,:,LEFT_HAND_IDXS] != 0).sum() / X_train[:,:,LEFT_HAND_IDXS].size / P_DATA_FILLED * 1e4\n# P_RIGHT_HAND_MEASUREMENTS = (X_train[:,:,RIGHT_HAND_IDXS] != 0).sum() / X_train[:,:,RIGHT_HAND_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_HAND_MEASUREMENTS: {P_LEFT_HAND_MEASUREMENTS:.2f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:19.400192Z","iopub.execute_input":"2024-12-05T23:24:19.400883Z","iopub.status.idle":"2024-12-05T23:24:25.901167Z","shell.execute_reply.started":"2024-12-05T23:24:19.400838Z","shell.execute_reply":"2024-12-05T23:24:25.900229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_left_right_hand_mean_std():\n    # LEFT HAND\n    LEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LEFT_HAND_IDXS], [2,3,0,1]).reshape([LEFT_HAND_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            # Plot\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Hands {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\n    LEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\n    \n    return LEFT_HANDS_MEAN, LEFT_HANDS_STD\n\nLEFT_HANDS_MEAN, LEFT_HANDS_STD = get_left_right_hand_mean_std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:25.902409Z","iopub.execute_input":"2024-12-05T23:24:25.902709Z","iopub.status.idle":"2024-12-05T23:24:35.471011Z","shell.execute_reply.started":"2024-12-05T23:24:25.902680Z","shell.execute_reply":"2024-12-05T23:24:35.470144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Statistics - Pose","metadata":{}},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_POSE_MEASUREMENTS = (X_train[:,:,POSE_IDXS] != 0).sum() / X_train[:,:,POSE_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_POSE_MEASUREMENTS: {P_POSE_MEASUREMENTS:.2f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:35.478036Z","iopub.execute_input":"2024-12-05T23:24:35.478328Z","iopub.status.idle":"2024-12-05T23:24:37.039268Z","shell.execute_reply.started":"2024-12-05T23:24:35.478298Z","shell.execute_reply":"2024-12-05T23:24:37.038318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_pose_mean_std():\n    # POSE\n    POSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                POSE_MEAN_X[col] = v.mean()\n                POSE_STD_X[col] = v.std()\n            if dim == 1: # Y\n                POSE_MEAN_Y[col] = v.mean()\n                POSE_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Pose {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    POSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\n    POSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T\n    \n    return POSE_MEAN, POSE_STD\n\nPOSE_MEAN, POSE_STD = get_pose_mean_std()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:37.040384Z","iopub.execute_input":"2024-12-05T23:24:37.040676Z","iopub.status.idle":"2024-12-05T23:24:39.544449Z","shell.execute_reply.started":"2024-12-05T23:24:37.040641Z","shell.execute_reply":"2024-12-05T23:24:39.543610Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Samples","metadata":{}},{"cell_type":"code","source":"# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            X_batch[i*n:(i+1)*n] = X[idxs]\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = NON_EMPTY_FRAME_IDXS[idxs]\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.545951Z","iopub.execute_input":"2024-12-05T23:24:39.546276Z","iopub.status.idle":"2024-12-05T23:24:39.552996Z","shell.execute_reply.started":"2024-12-05T23:24:39.546240Z","shell.execute_reply":"2024-12-05T23:24:39.551908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN)\nX_batch, y_batch = next(dummy_dataset)\n\nfor k, v in X_batch.items():\n    print(f'{k} shape: {v.shape}, dtype: {v.dtype}')\n\n# Batch shape/dtype\nprint(f'y_batch shape: {y_batch.shape}, dtype: {y_batch.dtype}')\n# Verify each batch contains each sign exactly N times\ndisplay(pd.Series(y_batch).value_counts().to_frame('Counts'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.554027Z","iopub.execute_input":"2024-12-05T23:24:39.554464Z","iopub.status.idle":"2024-12-05T23:24:39.633093Z","shell.execute_reply.started":"2024-12-05T23:24:39.554415Z","shell.execute_reply":"2024-12-05T23:24:39.632190Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Config","metadata":{}},{"cell_type":"code","source":"# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 512\n\n# Transformer\nNUM_BLOCKS = 2\nMLP_RATIO = 2\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu\n\nprint(f'UNITS: {UNITS}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.634356Z","iopub.execute_input":"2024-12-05T23:24:39.634735Z","iopub.status.idle":"2024-12-05T23:24:39.640607Z","shell.execute_reply.started":"2024-12-05T23:24:39.634693Z","shell.execute_reply":"2024-12-05T23:24:39.639714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transformer\n\nNeed to implement transformer from scratch as TFLite does not support the native TF implementation of MultiHeadAttention.","metadata":{}},{"cell_type":"code","source":"# based on: https://stackoverflow.com/questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.641670Z","iopub.execute_input":"2024-12-05T23:24:39.641983Z","iopub.status.idle":"2024-12-05T23:24:39.654166Z","shell.execute_reply.started":"2024-12-05T23:24:39.641957Z","shell.execute_reply":"2024-12-05T23:24:39.653260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Full Transformer\nclass Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for mha, mlp in zip(self.mhas, self.mlps):\n            x = x + mha(x, attention_mask)\n            x = x + mlp(x)\n    \n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.655260Z","iopub.execute_input":"2024-12-05T23:24:39.655601Z","iopub.status.idle":"2024-12-05T23:24:39.668095Z","shell.execute_reply.started":"2024-12-05T23:24:39.655562Z","shell.execute_reply":"2024-12-05T23:24:39.667346Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Landmark Embedding","metadata":{}},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.669079Z","iopub.execute_input":"2024-12-05T23:24:39.669445Z","iopub.status.idle":"2024-12-05T23:24:39.679456Z","shell.execute_reply.started":"2024-12-05T23:24:39.669417Z","shell.execute_reply":"2024-12-05T23:24:39.678567Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Embedding","metadata":{}},{"cell_type":"code","source":"class Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        # self.landmark_weights = tf.Variable(tf.zeros([3], dtype=tf.float32), name='landmark_weights')\n        self.landmark_weights = self.add_weight(name='landmark_weights',shape=[3],dtype=tf.float32,initializer=tf.zeros_initializer(),trainable=True)\n\n\n\n\n        \n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((\n            lips_embedding, left_hand_embedding, pose_embedding,\n        ), axis=3)\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        max_frame_idxs = tf.clip_by_value(\n                tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True),\n                1,\n                np.PINF,\n            )\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / max_frame_idxs * INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.680701Z","iopub.execute_input":"2024-12-05T23:24:39.681530Z","iopub.status.idle":"2024-12-05T23:24:39.694783Z","shell.execute_reply.started":"2024-12-05T23:24:39.681489Z","shell.execute_reply":"2024-12-05T23:24:39.693915Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Augmentation","metadata":{}},{"cell_type":"code","source":"# Not used, adds random X/y translation to input on samples level\nclass Augmentation(tf.keras.layers.Layer):\n    def __init__(self, noise_std):\n        super(Augmentation, self).__init__()\n        self.noise_std = noise_std\n    \n    def add_noise(self, t):\n        B = tf.shape(t)[0]\n        return tf.where(\n            t == 0.0,\n            0.0,\n            t + tf.random.normal([B,1,1,tf.shape(t)[3]], 0, self.noise_std),\n        )\n    \n    def call(self, lips0, left_hand0, pose0, training=False):\n        if training:\n            # Lips\n            lips0 = self.add_noise(lips0)\n            # Left Hand\n            left_hand0 = self.add_noise(left_hand0)\n            # Pose\n            pose0 = self.add_noise(pose0)\n        \n        return lips0, left_hand0, pose0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.695743Z","iopub.execute_input":"2024-12-05T23:24:39.695999Z","iopub.status.idle":"2024-12-05T23:24:39.708917Z","shell.execute_reply.started":"2024-12-05T23:24:39.695975Z","shell.execute_reply":"2024-12-05T23:24:39.708136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sparse Categorical Crossentropy With Label Smoothing","metadata":{}},{"cell_type":"code","source":"# source:: https://stackoverflow.com/questions/60689185/label-smoothing-for-sparse-categorical-crossentropy\ndef scce_with_ls(y_true, y_pred):\n    # One Hot Encode Sparsely Encoded Target Sign\n    y_true = tf.cast(y_true, tf.int32)\n    y_true = tf.one_hot(y_true, NUM_CLASSES, axis=1)\n    y_true = tf.squeeze(y_true, axis=2)\n    # Categorical Crossentropy with native label smoothing support\n    return tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.709862Z","iopub.execute_input":"2024-12-05T23:24:39.710135Z","iopub.status.idle":"2024-12-05T23:24:39.719862Z","shell.execute_reply.started":"2024-12-05T23:24:39.710106Z","shell.execute_reply":"2024-12-05T23:24:39.719099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# def get_model():\n#     # Inputs\n#     frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n#     non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n#     # Padding Mask\n#     mask0 = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n#     mask0 = tf.expand_dims(mask0, axis=2)\n#     # Random Frame Masking\n#     mask = tf.where(\n#         (tf.random.uniform(tf.shape(mask0)) > 0.25) & tf.math.not_equal(mask0, 0.0),\n#         1.0,\n#         0.0,\n#     )\n#     # Correct Samples Which are all masked now...\n#     mask = tf.where(\n#         tf.math.equal(tf.reduce_sum(mask, axis=[1,2], keepdims=True), 0.0),\n#         mask0,\n#         mask,\n#     )\n    \n    \n#     \"\"\"\n#         left_hand: 468:489\n#         pose: 489:522\n#         right_hand: 522:543\n#     \"\"\"\n#     x = frames\n#     x = tf.slice(x, [0,0,0,0], [-1,INPUT_SIZE, N_COLS, 2])\n#     # LIPS\n#     lips = tf.slice(x, [0,0,LIPS_START,0], [-1,INPUT_SIZE, 40, 2])\n#     lips = tf.where(\n#             tf.math.equal(lips, 0.0),\n#             0.0,\n#             (lips - LIPS_MEAN) / LIPS_STD,\n#         )\n#     # LEFT HAND\n#     left_hand = tf.slice(x, [0,0,40,0], [-1,INPUT_SIZE, 21, 2])\n#     left_hand = tf.where(\n#             tf.math.equal(left_hand, 0.0),\n#             0.0,\n#             (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n#         )\n#     # POSE\n#     pose = tf.slice(x, [0,0,61,0], [-1,INPUT_SIZE, 5, 2])\n#     pose = tf.where(\n#             tf.math.equal(pose, 0.0),\n#             0.0,\n#             (pose - POSE_MEAN) / POSE_STD,\n#         )\n    \n#     # Flatten\n#     lips = tf.reshape(lips, [-1, INPUT_SIZE, 40*2])\n#     left_hand = tf.reshape(left_hand, [-1, INPUT_SIZE, 21*2])\n#     pose = tf.reshape(pose, [-1, INPUT_SIZE, 5*2])\n        \n#     # Embedding\n#     x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n    \n#     # Encoder Transformer Blocks\n#     x = Transformer(NUM_BLOCKS)(x, mask)\n    \n#     # Pooling\n#     x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n#     # Classifier Dropout\n#     x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n#     # Classification Layer\n#     x = tf.keras.layers.Dense(NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    \n#     outputs = x\n    \n#     # Create Tensorflow Model\n#     model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n#     # Sparse Categorical Cross Entropy With Label Smoothing\n#     loss = scce_with_ls\n    \n#     # Adam Optimizer with weight decay\n#     optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    \n#     # TopK Metrics\n#     metrics = [\n#         tf.keras.metrics.SparseCategoricalAccuracy(name='acc'),\n#         tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name='top_5_acc'),\n#         tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name='top_10_acc'),\n#     ]\n    \n#     model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n#     return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.720921Z","iopub.execute_input":"2024-12-05T23:24:39.721175Z","iopub.status.idle":"2024-12-05T23:24:39.730275Z","shell.execute_reply.started":"2024-12-05T23:24:39.721149Z","shell.execute_reply":"2024-12-05T23:24:39.729623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\ndef get_model():\n    # 自定义切片层\n    class SliceLayer(tf.keras.layers.Layer):\n        def __init__(self, begin, size, **kwargs):\n            super(SliceLayer, self).__init__(**kwargs)\n            self.begin = begin\n            self.size = size\n\n        def call(self, inputs):\n            return tf.slice(inputs, self.begin, self.size)\n\n    # 自定义标准化层\n    class NormalizeLayer(tf.keras.layers.Layer):\n        def __init__(self, mean, std, **kwargs):\n            super(NormalizeLayer, self).__init__(**kwargs)\n            self.mean = mean\n            self.std = std\n\n        def call(self, inputs):\n            return tf.where(\n                tf.math.equal(inputs, 0.0),\n                0.0,\n                (inputs - self.mean) / self.std,\n            )\n\n    # 自定义 Reshape 层\n    class ReshapeLayer(tf.keras.layers.Layer):\n        def __init__(self, target_shape, **kwargs):\n            super(ReshapeLayer, self).__init__(**kwargs)\n            self.target_shape = target_shape\n\n        def call(self, inputs):\n            return tf.reshape(inputs, self.target_shape)\n\n    # 自定义 Padding Mask 层\n    class PaddingMaskLayer(tf.keras.layers.Layer):\n        def call(self, non_empty_frame_idxs):\n            mask = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n            return tf.expand_dims(mask, axis=2)\n\n    # 自定义随机帧掩码层\n    class RandomFrameMaskingLayer(tf.keras.layers.Layer):\n        def call(self, mask0):\n            mask = tf.where(\n                (tf.random.uniform(tf.shape(mask0)) > 0.25) & tf.math.not_equal(mask0, 0.0),\n                1.0,\n                0.0,\n            )\n            # 修正全被掩码的样本\n            mask = tf.where(\n                tf.math.equal(tf.reduce_sum(mask, axis=[1, 2], keepdims=True), 0.0),\n                mask0,\n                mask,\n            )\n            return mask\n\n    # 自定义 ReduceSum 层\n    class ReduceSumLayer(tf.keras.layers.Layer):\n        def __init__(self, axis, keepdims=False, **kwargs):\n            super(ReduceSumLayer, self).__init__(**kwargs)\n            self.axis = axis\n            self.keepdims = keepdims\n\n        def call(self, inputs):\n            return tf.reduce_sum(inputs, axis=self.axis, keepdims=self.keepdims)\n\n    # Inputs\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n\n    # Padding Mask\n    mask0 = PaddingMaskLayer()(non_empty_frame_idxs)\n    mask = RandomFrameMaskingLayer()(mask0)\n\n    \"\"\"\n        left_hand: 468:489\n        pose: 489:522\n        right_hand: 522:543\n    \"\"\"\n    x = frames\n    x = SliceLayer([0, 0, 0, 0], [-1, INPUT_SIZE, N_COLS, 2])(x)\n\n    # LIPS\n    lips = SliceLayer([0, 0, LIPS_START, 0], [-1, INPUT_SIZE, 40, 2])(x)\n    lips = NormalizeLayer(LIPS_MEAN, LIPS_STD)(lips)\n\n    # LEFT HAND\n    left_hand = SliceLayer([0, 0, 40, 0], [-1, INPUT_SIZE, 21, 2])(x)\n    left_hand = NormalizeLayer(LEFT_HANDS_MEAN, LEFT_HANDS_STD)(left_hand)\n\n    # POSE\n    pose = SliceLayer([0, 0, 61, 0], [-1, INPUT_SIZE, 5, 2])(x)\n    pose = NormalizeLayer(POSE_MEAN, POSE_STD)(pose)\n\n    # Flatten\n    lips = ReshapeLayer([-1, INPUT_SIZE, 40 * 2])(lips)\n    left_hand = ReshapeLayer([-1, INPUT_SIZE, 21 * 2])(left_hand)\n    pose = ReshapeLayer([-1, INPUT_SIZE, 5 * 2])(pose)\n\n    # Embedding\n    x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n\n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n\n    # Pooling (使用自定义 ReduceSumLayer)\n    pooled_sum = ReduceSumLayer(axis=1)(x * mask)\n    mask_sum = ReduceSumLayer(axis=1)(mask)\n    x = tf.keras.layers.Lambda(lambda inputs: inputs[0] / inputs[1])([pooled_sum, mask_sum])\n\n    # Classifier Dropout\n    x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n\n    # Classification Layer\n    x = tf.keras.layers.Dense(NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n\n    outputs = x\n\n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n\n    # Sparse Categorical Cross Entropy\n    loss = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=False)\n\n    # Adam Optimizer\n    # optimizer = tf.keras.optimizers.Adam(learning_rate=1e-3)\n    optimizer = tf.keras.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    # TopK Metrics\n    metrics = [\n        tf.keras.metrics.SparseCategoricalAccuracy(name='acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name='top_5_acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name='top_10_acc'),\n    ]\n\n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.731350Z","iopub.execute_input":"2024-12-05T23:24:39.731601Z","iopub.status.idle":"2024-12-05T23:24:39.749141Z","shell.execute_reply.started":"2024-12-05T23:24:39.731576Z","shell.execute_reply":"2024-12-05T23:24:39.748311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel = get_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:39.750216Z","iopub.execute_input":"2024-12-05T23:24:39.750557Z","iopub.status.idle":"2024-12-05T23:24:41.179797Z","shell.execute_reply.started":"2024-12-05T23:24:39.750519Z","shell.execute_reply":"2024-12-05T23:24:41.178822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot model summary\nmodel.summary(expand_nested=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:41.180902Z","iopub.execute_input":"2024-12-05T23:24:41.181179Z","iopub.status.idle":"2024-12-05T23:24:41.239509Z","shell.execute_reply.started":"2024-12-05T23:24:41.181151Z","shell.execute_reply":"2024-12-05T23:24:41.238681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:41.240785Z","iopub.execute_input":"2024-12-05T23:24:41.241138Z","iopub.status.idle":"2024-12-05T23:24:42.707754Z","shell.execute_reply.started":"2024-12-05T23:24:41.241099Z","shell.execute_reply":"2024-12-05T23:24:42.706836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# No NaN Predictions","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    y_pred = model.predict_on_batch(X_batch).flatten()\n\n    print(f'# NaN Values In Prediction: {np.isnan(y_pred).sum()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:42.708899Z","iopub.execute_input":"2024-12-05T23:24:42.709181Z","iopub.status.idle":"2024-12-05T23:24:47.153345Z","shell.execute_reply.started":"2024-12-05T23:24:42.709153Z","shell.execute_reply":"2024-12-05T23:24:47.152361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Weight Initialization","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    plt.figure(figsize=(12,5))\n    plt.title(f'Softmax Output Initialized Model | µ={y_pred.mean():.3f}, σ={y_pred.std():.3f}', pad=25)\n    pd.Series(y_pred).plot(kind='hist', bins=128, label='Class Probability')\n    plt.xlim(0, max(y_pred) * 1.1)\n    plt.vlines([1 / NUM_CLASSES], 0, plt.ylim()[1], color='red', label=f'Random Guessing Baseline 1/NUM_CLASSES={1 / NUM_CLASSES:.3f}')\n    plt.grid()\n    plt.legend()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:47.154437Z","iopub.execute_input":"2024-12-05T23:24:47.154713Z","iopub.status.idle":"2024-12-05T23:24:47.613428Z","shell.execute_reply.started":"2024-12-05T23:24:47.154685Z","shell.execute_reply":"2024-12-05T23:24:47.612527Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Learning Rate Scheduler","metadata":{}},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:47.614506Z","iopub.execute_input":"2024-12-05T23:24:47.614808Z","iopub.status.idle":"2024-12-05T23:24:47.620036Z","shell.execute_reply.started":"2024-12-05T23:24:47.614754Z","shell.execute_reply":"2024-12-05T23:24:47.619083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_lr_schedule(lr_schedule, epochs):\n    fig = plt.figure(figsize=(20, 10))\n    plt.plot([None] + lr_schedule + [None])\n    # X Labels\n    x = np.arange(1, epochs + 1)\n    x_axis_labels = [i if epochs <= 40 or i % 5 == 0 or i == 1 else None for i in range(1, epochs + 1)]\n    plt.xlim([1, epochs])\n    plt.xticks(x, x_axis_labels) # set tick step to 1 and let x axis start at 1\n    \n    # Increase y-limit for better readability\n    plt.ylim([0, max(lr_schedule) * 1.1])\n    \n    # Title\n    schedule_info = f'start: {lr_schedule[0]:.1E}, max: {max(lr_schedule):.1E}, final: {lr_schedule[-1]:.1E}'\n    plt.title(f'Step Learning Rate Schedule, {schedule_info}', size=18, pad=12)\n    \n    # Plot Learning Rates\n    for x, val in enumerate(lr_schedule):\n        if epochs <= 40 or x % 5 == 0 or x is epochs - 1:\n            if x < len(lr_schedule) - 1:\n                if lr_schedule[x - 1] < val:\n                    ha = 'right'\n                else:\n                    ha = 'left'\n            elif x == 0:\n                ha = 'right'\n            else:\n                ha = 'left'\n            plt.plot(x + 1, val, 'o', color='black');\n            offset_y = (max(lr_schedule) - min(lr_schedule)) * 0.02\n            plt.annotate(f'{val:.1E}', xy=(x + 1, val + offset_y), size=12, ha=ha)\n    \n    plt.xlabel('Epoch', size=16, labelpad=5)\n    plt.ylabel('Learning Rate', size=16, labelpad=5)\n    plt.grid()\n    plt.show()\n\n# Learning rate for encoder\nLR_SCHEDULE = [lrfn(step, num_warmup_steps=N_WARMUP_EPOCHS, lr_max=LR_MAX, num_cycles=0.50) for step in range(N_EPOCHS)]\n# Plot Learning Rate Schedule\nplot_lr_schedule(LR_SCHEDULE, epochs=N_EPOCHS)\n# Learning Rate Callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:47.621067Z","iopub.execute_input":"2024-12-05T23:24:47.621288Z","iopub.status.idle":"2024-12-05T23:24:47.940601Z","shell.execute_reply.started":"2024-12-05T23:24:47.621264Z","shell.execute_reply":"2024-12-05T23:24:47.939799Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Weight Decay Callback","metadata":{}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:47.941555Z","iopub.execute_input":"2024-12-05T23:24:47.941819Z","iopub.status.idle":"2024-12-05T23:24:47.946693Z","shell.execute_reply.started":"2024-12-05T23:24:47.941792Z","shell.execute_reply":"2024-12-05T23:24:47.945687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Performance Benchmark","metadata":{}},{"cell_type":"code","source":"%%timeit -n 100\nif TRAIN_MODEL:\n    # Verify model prediction is <<<100ms\n    model.predict_on_batch({ 'frames': X_train[:1], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_TRAIN[:1] })\n    pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:47.947620Z","iopub.execute_input":"2024-12-05T23:24:47.947887Z","iopub.status.idle":"2024-12-05T23:24:51.448073Z","shell.execute_reply.started":"2024-12-05T23:24:47.947860Z","shell.execute_reply":"2024-12-05T23:24:51.447148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Verify Validation Dataset Covers All Signs\n    print(f'# Unique Signs in Validation Set: {pd.Series(y_val).nunique()}')\n    # Value Counts\n    display(pd.Series(y_val).value_counts().to_frame('Count').iloc[[1,2,3,-3,-2,-1]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:51.449340Z","iopub.execute_input":"2024-12-05T23:24:51.450044Z","iopub.status.idle":"2024-12-05T23:24:51.460810Z","shell.execute_reply.started":"2024-12-05T23:24:51.449999Z","shell.execute_reply":"2024-12-05T23:24:51.459818Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate Initialzied Model","metadata":{}},{"cell_type":"code","source":"# Sanity Check\nif TRAIN_MODEL and USE_VAL:\n    _ = model.evaluate(*validation_data, verbose=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:51.461892Z","iopub.execute_input":"2024-12-05T23:24:51.462435Z","iopub.status.idle":"2024-12-05T23:24:59.751811Z","shell.execute_reply.started":"2024-12-05T23:24:51.462398Z","shell.execute_reply":"2024-12-05T23:24:59.751029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.753057Z","iopub.execute_input":"2024-12-05T23:24:59.753343Z","iopub.status.idle":"2024-12-05T23:24:59.761582Z","shell.execute_reply.started":"2024-12-05T23:24:59.753315Z","shell.execute_reply":"2024-12-05T23:24:59.757957Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"if TRAIN_MODEL:\n    # Clear all models in GPU\n    tf.keras.backend.clear_session()\n\n    # Get new fresh model\n    model = get_model()\n    \n    # Sanity Check\n    model.summary()\n\n    # Actual Training\n    history = model.fit(\n            x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN),\n            steps_per_epoch=len(X_train) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n            epochs=N_EPOCHS,\n            # Only used for validation data since training data is a generator\n            batch_size=BATCH_SIZE,\n            validation_data=validation_data,\n            callbacks=[\n                lr_callback,\n                WeightDecayCallback(),\n            ],\n            verbose = VERBOSE,\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.762215Z","iopub.status.idle":"2024-12-05T23:24:59.762752Z","shell.execute_reply.started":"2024-12-05T23:24:59.762562Z","shell.execute_reply":"2024-12-05T23:24:59.762589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save Model Weights\nmodel.save_weights('model.weights.h5')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.764826Z","iopub.status.idle":"2024-12-05T23:24:59.765307Z","shell.execute_reply.started":"2024-12-05T23:24:59.765060Z","shell.execute_reply":"2024-12-05T23:24:59.765086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if USE_VAL:\n    # Validation Predictions\n    y_val_pred = model.predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2).argmax(axis=1)\n    # Label\n    labels = [ORD2SIGN.get(i).replace(' ', '_') for i in range(NUM_CLASSES)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.767457Z","iopub.status.idle":"2024-12-05T23:24:59.767954Z","shell.execute_reply.started":"2024-12-05T23:24:59.767685Z","shell.execute_reply":"2024-12-05T23:24:59.767713Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Landmark Attention Weights","metadata":{}},{"cell_type":"code","source":"[w.name for w in model.get_layer('embedding').weights]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.769484Z","iopub.status.idle":"2024-12-05T23:24:59.769989Z","shell.execute_reply.started":"2024-12-05T23:24:59.769706Z","shell.execute_reply":"2024-12-05T23:24:59.769731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Landmark Weights\nfor w in model.get_layer('embedding').weights:\n    if 'landmark_weights' in w.name:\n        weights = scipy.special.softmax(w)\n\nlandmarks = ['lips_embedding', 'left_hand_embedding', 'pose_embedding']\n\nfor w, lm in zip(weights, landmarks):\n    print(f'{lm} weight: {(w*100):.1f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.771082Z","iopub.status.idle":"2024-12-05T23:24:59.771537Z","shell.execute_reply.started":"2024-12-05T23:24:59.771295Z","shell.execute_reply":"2024-12-05T23:24:59.771321Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Classification Report","metadata":{}},{"cell_type":"code","source":"def print_classification_report():\n    # Classification report for all signs\n    classification_report = sklearn.metrics.classification_report(\n            y_val,\n            y_val_pred,\n            target_names=labels,\n            output_dict=True,\n        )\n    # Round Data for better readability\n    classification_report = pd.DataFrame(classification_report).T\n    classification_report = classification_report.round(2)\n    classification_report = classification_report.astype({\n            'support': np.uint16,\n        })\n    # Add signs\n    classification_report['sign'] = [e if e in SIGN2ORD else -1 for e in classification_report.index]\n    classification_report['sign_ord'] = classification_report['sign'].apply(SIGN2ORD.get).fillna(-1).astype(np.int16)\n    # Sort on F1-score\n    classification_report = pd.concat((\n        classification_report.head(NUM_CLASSES).sort_values('f1-score', ascending=False),\n        classification_report.tail(3),\n    ))\n\n    pd.options.display.max_rows = 999\n    display(classification_report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.772631Z","iopub.status.idle":"2024-12-05T23:24:59.773099Z","shell.execute_reply.started":"2024-12-05T23:24:59.772863Z","shell.execute_reply":"2024-12-05T23:24:59.772888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if USE_VAL:\n    print_classification_report()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.774570Z","iopub.status.idle":"2024-12-05T23:24:59.774923Z","shell.execute_reply.started":"2024-12-05T23:24:59.774735Z","shell.execute_reply":"2024-12-05T23:24:59.774752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training History","metadata":{}},{"cell_type":"code","source":"def plot_history_metric(metric, f_best=np.argmax, ylim=None, yscale=None, yticks=None):\n    plt.figure(figsize=(20, 10))\n    \n    values = history.history[metric]\n    N_EPOCHS = len(values)\n    val = 'val' in ''.join(history.history.keys())\n    # Epoch Ticks\n    if N_EPOCHS <= 20:\n        x = np.arange(1, N_EPOCHS + 1)\n    else:\n        x = [1, 5] + [10 + 5 * idx for idx in range((N_EPOCHS - 10) // 5 + 1)]\n\n    x_ticks = np.arange(1, N_EPOCHS+1)\n\n    # Validation\n    if val:\n        val_values = history.history[f'val_{metric}']\n        val_argmin = f_best(val_values)\n        plt.plot(x_ticks, val_values, label=f'val')\n\n    # summarize history for accuracy\n    plt.plot(x_ticks, values, label=f'train')\n    argmin = f_best(values)\n    plt.scatter(argmin + 1, values[argmin], color='red', s=75, marker='o', label=f'train_best')\n    if val:\n        plt.scatter(val_argmin + 1, val_values[val_argmin], color='purple', s=75, marker='o', label=f'val_best')\n\n    plt.title(f'Model {metric}', fontsize=24, pad=10)\n    plt.ylabel(metric, fontsize=20, labelpad=10)\n\n    if ylim:\n        plt.ylim(ylim)\n\n    if yscale is not None:\n        plt.yscale(yscale)\n        \n    if yticks is not None:\n        plt.yticks(yticks, fontsize=16)\n\n    plt.xlabel('epoch', fontsize=20, labelpad=10)        \n    plt.tick_params(axis='x', labelsize=8)\n    plt.xticks(x, fontsize=16) # set tick step to 1 and let x axis start at 1\n    plt.yticks(fontsize=16)\n    \n    plt.legend(prop={'size': 10})\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.776245Z","iopub.status.idle":"2024-12-05T23:24:59.776701Z","shell.execute_reply.started":"2024-12-05T23:24:59.776455Z","shell.execute_reply":"2024-12-05T23:24:59.776479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('loss', f_best=np.argmin)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.778004Z","iopub.status.idle":"2024-12-05T23:24:59.778441Z","shell.execute_reply.started":"2024-12-05T23:24:59.778213Z","shell.execute_reply":"2024-12-05T23:24:59.778236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.779503Z","iopub.status.idle":"2024-12-05T23:24:59.779988Z","shell.execute_reply.started":"2024-12-05T23:24:59.779734Z","shell.execute_reply":"2024-12-05T23:24:59.779774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_5_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.781203Z","iopub.status.idle":"2024-12-05T23:24:59.781662Z","shell.execute_reply.started":"2024-12-05T23:24:59.781416Z","shell.execute_reply":"2024-12-05T23:24:59.781441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_10_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.783075Z","iopub.status.idle":"2024-12-05T23:24:59.783523Z","shell.execute_reply.started":"2024-12-05T23:24:59.783287Z","shell.execute_reply":"2024-12-05T23:24:59.783311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission\n\nSubmission code loosley based on [this notebook](https://www.kaggle.com/code/dschettler8845/gislr-learn-eda-baseline#baseline) by [Darien Schettler\n](https://www.kaggle.com/dschettler8845)","metadata":{}},{"cell_type":"code","source":"# # TFLite model for submission\n# class TFLiteModel(tf.Module):\n#     def __init__(self, model):\n#         super(TFLiteModel, self).__init__()\n\n#         # Load the feature generation and main models\n#         self.preprocess_layer = preprocess_layer\n#         self.model = model\n    \n#     @tf.function(input_signature=[tf.TensorSpec(shape=[None, N_ROWS, N_DIMS], dtype=tf.float32, name='inputs')])\n#     def __call__(self, inputs):\n#         # Preprocess Data\n#         x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n#         # Add Batch Dimension\n#         x = tf.expand_dims(x, axis=0)\n#         non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n#         # Make Prediction\n#         outputs = self.model({ 'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs })\n#         # Squeeze Output 1x250 -> 250\n#         outputs = tf.squeeze(outputs, axis=0)\n\n#         # Return a dictionary with the output tensor\n#         return {'outputs': outputs}\n\n# # Define TF Lite Model\n# tflite_keras_model = TFLiteModel(model)\n# i=5\n# # Sanity Check\n# demo_raw_data = load_relevant_data_subset(train['file_path'].values[i])\n# print(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\n# demo_output = tflite_keras_model(demo_raw_data)[\"outputs\"]\n# print(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\n# demo_prediction = demo_output.numpy().argmax()\n# print(f'demo_prediction: {demo_prediction}, correct: {train.iloc[i][\"sign_ord\"]}')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.784479Z","iopub.status.idle":"2024-12-05T23:24:59.784944Z","shell.execute_reply.started":"2024-12-05T23:24:59.784686Z","shell.execute_reply":"2024-12-05T23:24:59.784709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model_test = get_model()\n# model_test.load_weights('/kaggle/input/islr-model-50epoch-kaggle/model241202.weights.h5')\n# model_test.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.786443Z","iopub.status.idle":"2024-12-05T23:24:59.786734Z","shell.execute_reply.started":"2024-12-05T23:24:59.786584Z","shell.execute_reply":"2024-12-05T23:24:59.786598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow.keras import layers, Model\n\n# # Assuming preprocess_layer, and model are already defined, \n# # you should refactor everything into a Keras model\n# class CustomTFLiteModel(Model):\n#     def __init__(self, preprocess_layer, main_model):\n#         super(CustomTFLiteModel, self).__init__()\n#         self.preprocess_layer = preprocess_layer\n#         self.main_model = main_model\n\n#     def call(self, inputs):\n#         # Preprocess the input\n#         x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n#         # Add Batch Dimension\n#         x = tf.expand_dims(x, axis=0)\n#         non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n#         # Make Prediction\n#         outputs = self.main_model({'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs})\n#         # Squeeze Output 1x250 -> 250\n#         outputs = tf.squeeze(outputs, axis=0)\n#         return outputs\n\n# # Create the custom Keras model instance\n# custom_tflite_keras_model = CustomTFLiteModel(preprocess_layer, model_test)\n# i=1\n# # Test the model with some input data to ensure it works\n# demo_raw_data = load_relevant_data_subset(train['file_path'].values[i])\n# print(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\n# demo_output = custom_tflite_keras_model(demo_raw_data)\n# print(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\n# demo_prediction = demo_output.numpy().argmax()\n# print(f'demo_prediction: {demo_prediction}, correct: {train.iloc[i][\"sign_ord\"]}')\n\n# print(\"output gloss: \",ORD2SIGN[demo_prediction])\n# print(train.iloc[i])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.787631Z","iopub.status.idle":"2024-12-05T23:24:59.787956Z","shell.execute_reply.started":"2024-12-05T23:24:59.787803Z","shell.execute_reply":"2024-12-05T23:24:59.787820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 批量处理MP4到Parquet文件","metadata":{}},{"cell_type":"code","source":"# import os\n# import cv2\n# import mediapipe as mp\n# import pandas as pd\n# import numpy as np\n\n# mp_holistic = mp.solutions.holistic\n\n# def process_video_to_parquet(video_path, output_parquet_path):\n#     \"\"\"\n#     处理单个视频，将 Mediapipe 的关键点保存为 Parquet 格式。\n#     \"\"\"\n#     cap = cv2.VideoCapture(video_path)\n#     holistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.1)\n\n#     video_data = []\n#     frame_no = 0\n\n#     while cap.isOpened():\n#         print('\\rProcessing frame:', frame_no, end='')\n#         success, image = cap.read()\n#         if not success:\n#             break\n\n#         image = cv2.resize(image, dsize=None, fx=4, fy=4)\n#         height, width, _ = image.shape\n#         fy = height / width\n\n#         image.flags.writeable = False\n#         image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#         result = holistic.process(image)\n\n#         data = []\n\n#         # Face landmarks\n#         if result.face_landmarks is None:\n#             for i in range(468):\n#                 data.append({'type': 'face', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n#         else:\n#             for i, landmark in enumerate(result.face_landmarks.landmark):\n#                 data.append({'type': 'face', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n#         # Left hand landmarks\n#         if result.left_hand_landmarks is None:\n#             for i in range(21):\n#                 data.append({'type': 'left_hand', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n#         else:\n#             for i, landmark in enumerate(result.left_hand_landmarks.landmark):\n#                 data.append({'type': 'left_hand', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n#         # Pose landmarks\n#         if result.pose_landmarks is None:\n#             for i in range(33):\n#                 data.append({'type': 'pose', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n#         else:\n#             for i, landmark in enumerate(result.pose_landmarks.landmark):\n#                 data.append({'type': 'pose', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n#         # Right hand landmarks\n#         if result.right_hand_landmarks is None:\n#             for i in range(21):\n#                 data.append({'type': 'right_hand', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n#         else:\n#             for i, landmark in enumerate(result.right_hand_landmarks.landmark):\n#                 data.append({'type': 'right_hand', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n#         frame_df = pd.DataFrame(data)\n#         frame_df['frame'] = frame_no\n#         frame_df['height'] = height / width\n#         frame_df['width'] = width / width\n#         video_data.append(frame_df)\n\n#         frame_no += 1\n\n#     cap.release()\n#     holistic.close()\n\n#     if video_data:\n#         video_df = pd.concat(video_data, ignore_index=True)\n#         video_df.to_parquet(output_parquet_path, index=False)\n#         print(f\"\\nParquet file saved at: {output_parquet_path}\")\n#     else:\n#         print(f\"\\nNo data extracted for: {video_path}\")\n\n# def process_videos_to_parquet(input_dir, output_dir):\n#     \"\"\"\n#     批量处理输入文件夹中的视频，并输出 Parquet 文件。\n#     \"\"\"\n#     if not os.path.exists(output_dir):\n#         os.makedirs(output_dir)\n\n#     for video_file in os.listdir(input_dir):\n#         if video_file.endswith((\".mp4\", \".avi\")):\n#             input_video_path = os.path.join(input_dir, video_file)\n#             output_parquet_path = os.path.join(output_dir, f\"{os.path.splitext(video_file)[0]}.parquet\")\n#             print(f\"Processing: {video_file}\")\n#             process_video_to_parquet(input_video_path, output_parquet_path)\n            \n# # 主函数：设置输入视频路径和输出路径\n# input_videos_dir = \"/kaggle/input/sign2gloss-validation/car\"  # 替换为存放视频的文件夹路径\n# output_parquet_dir = \"/kaggle/working/output_car/parquets\"  # 替换为存放 Parquet 文件的文件夹路径\n\n# process_videos_to_parquet(input_videos_dir, output_parquet_dir)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.788784Z","iopub.status.idle":"2024-12-05T23:24:59.789104Z","shell.execute_reply.started":"2024-12-05T23:24:59.788956Z","shell.execute_reply":"2024-12-05T23:24:59.788972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n\n# def get_all_file_paths(directory):\n#     \"\"\"\n#     获取给定路径下的所有文件的路径列表\n#     :param directory: str，目录路径\n#     :return: list，包含所有文件路径的列表\n#     \"\"\"\n#     file_paths = []\n#     for root, dirs, files in os.walk(directory):\n#         for file in files:\n#             file_paths.append(os.path.join(root, file))\n#     return file_paths\n\n# # 使用示例\n# directory_path = \"/kaggle/working/output_car/parquets\"  # 替换为你想要遍历的目录路径\n# all_files = get_all_file_paths(directory_path)\n# print(all_files)\n\n# def predict_top5_from_list(paths, model, ord2sign):\n#     \"\"\"\n#     Perform inference on a list of file paths and return Top-5 predictions for each.\n\n#     Args:\n#         paths (list): List of file paths to process.\n#         model (tf.keras.Model): The pre-trained custom model.\n#         ord2sign (dict): Mapping from indices to gloss labels.\n\n#     Returns:\n#         dict: A dictionary where keys are file paths and values are lists of Top-5 gloss predictions.\n#     \"\"\"\n#     results = {}\n#     for path in paths:\n#         # 加载和预处理输入数据\n#         demo_raw_data = load_relevant_data_subset(path)\n        \n#         # 推理\n#         demo_output = model(demo_raw_data)\n\n#         # 提取 Top-5 预测结果\n#         top_k = 5\n#         top5_predictions = tf.math.top_k(demo_output, k=top_k)\n#         top5_indices = top5_predictions.indices.numpy()\n#         top5_probabilities = top5_predictions.values.numpy()\n\n#         # 将索引映射到 gloss 标签\n#         top5_glosses = [ord2sign[idx] for idx in top5_indices]\n\n#         # 保存结果\n#         results[path] = {\n#             \"top5_glosses\": top5_glosses,\n#             \"top5_probabilities\": top5_probabilities.tolist(),\n#         }\n\n#     return results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.790638Z","iopub.status.idle":"2024-12-05T23:24:59.790964Z","shell.execute_reply.started":"2024-12-05T23:24:59.790813Z","shell.execute_reply":"2024-12-05T23:24:59.790830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def process_model_results_with_files(all_files, results):\n#     \"\"\"\n#     Process model inference results into a structured Pandas DataFrame with file paths included.\n    \n#     Parameters:\n#     - all_files (list): List of file paths corresponding to the input files.\n#     - results (dict): Dictionary containing model inference results with keys as indexes and values\n#                       as dictionaries with 'top5_glosses' and 'top5_probabilities'.\n    \n#     Returns:\n#     - pd.DataFrame: A DataFrame with columns ['FilePath', 'Top1', 'Top2', 'Top3', 'Top4', 'Top5'].\n#     \"\"\"\n#     # Ensure the length of all_files matches the number of results\n#     if len(all_files) != len(results):\n#         raise ValueError(\"Length of all_files does not match the number of results.\")\n    \n#     # Create a structured table with FilePath and gloss-prob formatted values\n#     data = []\n#     for file_path, (i, result) in zip(all_files, results.items()):\n#         row = [file_path] + [\n#             f\"{gloss}-{prob:.2f}\" for gloss, prob in zip(result['top5_glosses'], result['top5_probabilities'])\n#         ]\n#         data.append(row)\n    \n#     # Convert to DataFrame with appropriate columns\n#     df = pd.DataFrame(data, columns=[\"FilePath\", \"Top1\", \"Top2\", \"Top3\", \"Top4\", \"Top5\"])\n#     return df\n# res = predict_top5_from_list(all_files, custom_tflite_keras_model, ORD2SIGN)\n\n# # 调用新函数，生成 DataFrame\n# df = process_model_results_with_files(all_files, res)\n# print(get_all_file_paths(\"/kaggle/input/sign2gloss-validation/car\"))\n# # 打印结果或进一步处理\n# df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.791941Z","iopub.status.idle":"2024-12-05T23:24:59.792247Z","shell.execute_reply.started":"2024-12-05T23:24:59.792098Z","shell.execute_reply":"2024-12-05T23:24:59.792114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport mediapipe as mp\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\n\n\n\n\n\n\n\nmp_holistic = mp.solutions.holistic\n\nmodel_test = get_model()\nmodel_test.load_weights('/kaggle/input/islr-model-50epoch-kaggle/model241202.weights.h5')\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Model\n\n# Assuming preprocess_layer, and model are already defined, \n# you should refactor everything into a Keras model\nclass CustomTFLiteModel(Model):\n    def __init__(self, preprocess_layer, main_model):\n        super(CustomTFLiteModel, self).__init__()\n        self.preprocess_layer = preprocess_layer\n        self.main_model = main_model\n\n    def call(self, inputs):\n        # Preprocess the input\n        x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n        # Add Batch Dimension\n        x = tf.expand_dims(x, axis=0)\n        non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n        # Make Prediction\n        outputs = self.main_model({'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs})\n        # Squeeze Output 1x250 -> 250\n        outputs = tf.squeeze(outputs, axis=0)\n        return outputs\n\n# Create the custom Keras model instance\ncustom_tflite_keras_model = CustomTFLiteModel(preprocess_layer, model_test)\n\n\n\n\ndef process_video_to_parquet(video_path, output_parquet_path):\n    \"\"\"\n    处理单个视频，将 Mediapipe 的关键点保存为 Parquet 格式。\n    \"\"\"\n\n\n\n\n\n    \n    cap = cv2.VideoCapture(video_path)\n    holistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.1)\n\n    video_data = []\n    frame_no = 0\n\n    while cap.isOpened():\n        # print('\\rProcessing frame:', frame_no, end='')\n        success, image = cap.read()\n        if not success:\n            break\n\n        image = cv2.resize(image, dsize=None, fx=4, fy=4)\n        height, width, _ = image.shape\n        fy = height / width\n\n        image.flags.writeable = False\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        result = holistic.process(image)\n\n        data = []\n\n        # Face landmarks\n        if result.face_landmarks is None:\n            for i in range(468):\n                data.append({'type': 'face', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n        else:\n            for i, landmark in enumerate(result.face_landmarks.landmark):\n                data.append({'type': 'face', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n        # Left hand landmarks\n        if result.left_hand_landmarks is None:\n            for i in range(21):\n                data.append({'type': 'left_hand', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n        else:\n            for i, landmark in enumerate(result.left_hand_landmarks.landmark):\n                data.append({'type': 'left_hand', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n        # Pose landmarks\n        if result.pose_landmarks is None:\n            for i in range(33):\n                data.append({'type': 'pose', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n        else:\n            for i, landmark in enumerate(result.pose_landmarks.landmark):\n                data.append({'type': 'pose', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n        # Right hand landmarks\n        if result.right_hand_landmarks is None:\n            for i in range(21):\n                data.append({'type': 'right_hand', 'landmark_index': i, 'x': np.nan, 'y': np.nan, 'z': np.nan})\n        else:\n            for i, landmark in enumerate(result.right_hand_landmarks.landmark):\n                data.append({'type': 'right_hand', 'landmark_index': i, 'x': landmark.x, 'y': landmark.y * fy, 'z': landmark.z})\n\n        frame_df = pd.DataFrame(data)\n        frame_df['frame'] = frame_no\n        frame_df['height'] = height / width\n        frame_df['width'] = width / width\n        video_data.append(frame_df)\n\n        frame_no += 1\n\n    cap.release()\n    holistic.close()\n\n    if video_data:\n        video_df = pd.concat(video_data, ignore_index=True)\n        video_df.to_parquet(output_parquet_path, index=False)\n        # print(f\"\\nParquet file saved at: {output_parquet_path}\")\n    else:\n        print(f\"\\nNo data extracted for: {video_path}\")\n\ndef process_videos_to_parquet(input_dir, output_dir):\n    \"\"\"\n    批量处理输入文件夹中的视频，并输出 Parquet 文件。\n    \"\"\"\n    if not os.path.exists(output_dir):\n        os.makedirs(output_dir)\n\n    for video_file in os.listdir(input_dir):\n        if video_file.endswith((\".mp4\", \".avi\")):\n            input_video_path = os.path.join(input_dir, video_file)\n            output_parquet_path = os.path.join(output_dir, f\"{os.path.splitext(video_file)[0]}.parquet\")\n            # print(f\"Processing: {video_file}\")\n            process_video_to_parquet(input_video_path, output_parquet_path)\n\ndef get_all_file_paths(directory):\n    \"\"\"\n    获取给定路径下的所有文件的路径列表\n    :param directory: str，目录路径\n    :return: list，包含所有文件路径的列表\n    \"\"\"\n    file_paths = []\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            file_paths.append(os.path.join(root, file))\n    return file_paths\n\ndef is_frontal_face(face_landmarks):\n    \"\"\"\n    判断面部是否正对镜头。\n    :param face_landmarks: 面部关键点列表 (468, 3)\n    :return: True 如果面部正对镜头，否则 False\n    \"\"\"\n    if face_landmarks is None:\n        return False\n\n    # 提取鼻尖与眼睛的关键点\n    nose = face_landmarks[1]\n    left_eye = face_landmarks[33]\n    right_eye = face_landmarks[263]\n\n    # 判断鼻尖是否接近眼睛的中间点\n    center_x = (left_eye.x + right_eye.x) / 2\n    center_y = (left_eye.y + right_eye.y) / 2\n    nose_to_center_distance = np.sqrt((nose.x - center_x)**2 + (nose.y - center_y)**2)\n\n    # 如果鼻尖偏离中心点太多，则面部不是正面\n    return nose_to_center_distance < 0.05  # 阈值可调\n\ndef has_active_hand(left_hand, right_hand, active_threshold=0.01):\n    \"\"\"\n    判断是否仅有一个活跃的手。\n    :param left_hand: 左手关键点\n    :param right_hand: 右手关键点\n    :param active_threshold: 活跃手的阈值\n    :return: True 如果仅有一个手活跃，否则 False\n    \"\"\"\n    # 计算左手的活动范围\n    if left_hand:\n        left_coords = np.array([[landmark.x, landmark.y, landmark.z] for landmark in left_hand])\n        left_range = np.nanstd(left_coords)\n    else:\n        left_range = 0\n\n    # 计算右手的活动范围\n    if right_hand:\n        right_coords = np.array([[landmark.x, landmark.y, landmark.z] for landmark in right_hand])\n        right_range = np.nanstd(right_coords)\n    else:\n        right_range = 0\n\n    # 判断单手活跃性\n    is_left_active = left_range > active_threshold\n    is_right_active = right_range > active_threshold\n\n    # 如果仅有一只手活跃，则返回 True\n    return (is_left_active and not is_right_active) or (is_right_active and not is_left_active)\n\ndef process_and_filter_video(video_path, require_frontal_face=True, require_active_hands=False, valid_frame_ratio=0.8):\n    \"\"\"\n    判断单个视频是否符合筛选条件。\n    :param video_path: 输入视频路径\n    :param require_frontal_face: 是否需要正面朝向摄像机（布尔值）\n    :param require_active_hands: 是否需要至少一只活跃的手（布尔值）\n    :param valid_frame_ratio: 有效帧占总帧数的最低比例（默认0.8）\n    :return: True 如果视频符合条件，否则 False\n    \"\"\"\n    # print(f\"Processing for quality check: {video_path}\")\n    cap = cv2.VideoCapture(video_path)\n    holistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.1)\n    \n    valid_frames = 0\n    total_frames = 0\n\n    while cap.isOpened():\n        success, image = cap.read()\n        if not success:\n            break\n\n        total_frames += 1\n        image = cv2.resize(image, dsize=None, fx=4, fy=4)\n        image.flags.writeable = False\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        # Mediapipe 处理\n        result = holistic.process(image)\n\n        # 判断是否符合筛选标准\n        is_valid_frame = True\n\n        if require_frontal_face:\n            is_valid_frame = is_valid_frame and is_frontal_face(result.face_landmarks.landmark if result.face_landmarks else None)\n\n        if require_active_hands:\n            is_valid_frame = is_valid_frame and has_active_hand(\n                result.left_hand_landmarks.landmark if result.left_hand_landmarks else None,\n                result.right_hand_landmarks.landmark if result.right_hand_landmarks else None\n            )\n\n        if is_valid_frame:\n            valid_frames += 1\n\n    cap.release()\n    holistic.close()\n\n    # 判断是否满足有效帧比例\n    if total_frames > 0 and valid_frames / total_frames >= valid_frame_ratio:\n        # print(f\"Video {video_path} passed the quality check.\")\n        return True\n    else:\n        # print(f\"Video {video_path} failed the quality check.\")\n        return False\n        \ndef predict_top5_from_list(paths, model, ord2sign):\n    \"\"\"\n    Perform inference on a list of file paths and return Top-5 predictions for each.\n\n    Args:\n        paths (list): List of file paths to process.\n        model (tf.keras.Model): The pre-trained custom model.\n        ord2sign (dict): Mapping from indices to gloss labels.\n\n    Returns:\n        dict: A dictionary where keys are file paths and values are lists of Top-5 gloss predictions.\n    \"\"\"\n    results = {}\n    for path in paths:\n        # 加载和预处理输入数据\n        demo_raw_data = load_relevant_data_subset(path)\n        \n        # 推理\n        demo_output = model(demo_raw_data)\n\n        # 提取 Top-5 预测结果\n        top_k = 5\n        top5_predictions = tf.math.top_k(demo_output, k=top_k)\n        top5_indices = top5_predictions.indices.numpy()\n        top5_probabilities = top5_predictions.values.numpy()\n\n        # 将索引映射到 gloss 标签\n        top5_glosses = [ord2sign[idx] for idx in top5_indices]\n\n        # 保存结果\n        results[path] = {\n            \"top5_glosses\": top5_glosses,\n            \"top5_probabilities\": top5_probabilities.tolist(),\n        }\n\n    return results\n\n\n\ndef main(mp4_dir, output_parquet_dir, model, ord2sign):\n    # 步骤1：处理视频并生成 Parquet 文件\n    process_videos_to_parquet(mp4_dir, output_parquet_dir)\n\n    # 获取 MP4 文件列表\n    mp4_files = [f for f in os.listdir(mp4_dir) if f.endswith(('.mp4', '.avi'))]\n\n    data = []\n\n    for video_file in mp4_files:\n        mp4_file_path = os.path.join(mp4_dir, video_file)\n        parquet_file_name = os.path.splitext(video_file)[0] + '.parquet'\n        parquet_file_path = os.path.join(output_parquet_dir, parquet_file_name)\n\n        # 数据质量检测结果1（正面）\n        result_frontal_face = process_and_filter_video(mp4_file_path, require_frontal_face=True, require_active_hands=False)\n\n        # 数据质量检测结果2（双手）\n        result_active_hands = process_and_filter_video(mp4_file_path, require_frontal_face=False, require_active_hands=True)\n\n        data.append({\n            'mp4_file_path': mp4_file_path,\n            'parquet_file_path': parquet_file_path,\n            'frontal_face': result_frontal_face,\n            'active_hands': result_active_hands\n        })\n\n    # 获取所有 Parquet 文件路径\n    all_parquet_files = [item['parquet_file_path'] for item in data]\n\n    # 步骤2：模型预测\n    predictions = predict_top5_from_list(all_parquet_files, model, ord2sign)\n\n    # 将预测结果添加到数据中\n    for item in data:\n        parquet_file_path = item['parquet_file_path']\n        if parquet_file_path in predictions:\n            prediction = predictions[parquet_file_path]\n            top5_glosses = prediction['top5_glosses']\n            top5_probabilities = prediction['top5_probabilities']\n            for i in range(5):\n                item[f'Top{i+1}'] = f\"{top5_glosses[i]}-{top5_probabilities[i]:.2f}\"\n        else:\n            for i in range(5):\n                item[f'Top{i+1}'] = None\n\n    # 创建最终的 DataFrame\n    df = pd.DataFrame(data, columns=[\n        'mp4_file_path', 'parquet_file_path', 'Top1', 'Top2', 'Top3', 'Top4', 'Top5', 'frontal_face', 'active_hands'\n    ])\n\n    return df\n\n\nmp4_dir = \"/kaggle/input/sign2gloss-validation/car\"  # 替换为您的 MP4 文件夹路径\n# 输出 Parquet 文件夹路径\noutput_parquet_dir = \"/kaggle/working/output_car/parquets\"  # 替换为您的输出 Parquet 文件夹路径\n\n\nmodel = custom_tflite_keras_model\n\n # 请在此处加载您的索引到手语词汇的映射字典\n# 例如：ord2sign = {0: 'hello', 1: 'thank_you', ...}\nord2sign =ORD2SIGN\n\n# 调用主函数\ndf = main(mp4_dir, output_parquet_dir, model, ord2sign)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:26:00.114067Z","iopub.execute_input":"2024-12-05T23:26:00.114459Z","iopub.status.idle":"2024-12-05T23:28:23.080255Z","shell.execute_reply.started":"2024-12-05T23:26:00.114424Z","shell.execute_reply":"2024-12-05T23:28:23.079466Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:28:23.082227Z","iopub.execute_input":"2024-12-05T23:28:23.082980Z","iopub.status.idle":"2024-12-05T23:28:23.096019Z","shell.execute_reply.started":"2024-12-05T23:28:23.082932Z","shell.execute_reply":"2024-12-05T23:28:23.094954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom tqdm import tqdm  # 用于显示进度条\n\n# 定义路径\nbase_input_dir = '/kaggle/input/sign2gloss-validation'\nbase_output_dir = '/kaggle/working/output'\n\n# 假设以下变量已经定义\nmodel = custom_tflite_keras_model\nord2sign = ORD2SIGN\n\n# 初始化一个空的 DataFrame 用于拼接\nfinal_df = pd.DataFrame()\n\n# 获取所有子文件夹\nsubfolders = [f.path for f in os.scandir(base_input_dir) if f.is_dir()]\n\n# 遍历每个子文件夹并显示进度条\nwith tqdm(subfolders, desc=\"Processing gloss folders\", unit=\"folder\") as pbar:\n    for subfolder in pbar:\n        # 提取当前的 gloss 名称\n        gloss = os.path.basename(subfolder)\n        \n        # 动态更新进度条描述\n        pbar.set_description(f\"Processing gloss {gloss}\")\n        \n        # 定义 MP4 输入文件夹路径\n        mp4_dir = subfolder\n        \n        # 动态生成输出 Parquet 文件夹路径\n        output_parquet_dir = f\"{base_output_dir}_{gloss}/parquets\"\n        os.makedirs(output_parquet_dir, exist_ok=True)  # 确保输出文件夹存在\n\n        # 调用主函数处理数据\n        df = main(mp4_dir, output_parquet_dir, model, ord2sign)\n        \n        # 添加 gloss 列\n        df['gloss'] = gloss\n        \n        # 垂直拼接结果\n        final_df = pd.concat([final_df, df], ignore_index=True)\n\n# 保存最终结果为 CSV 或 Parquet 文件\nfinal_df.to_csv('/kaggle/working/final_output.csv', index=False)\nprint(\"最终结果已保存为 /kaggle/working/final_output.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:28:23.097173Z","iopub.execute_input":"2024-12-05T23:28:23.097458Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 导出模型为 TensorFlow SavedModel 格式\n# saved_model_path = \"/kaggle/working/saved_model\"\n# custom_tflite_keras_model.export(saved_model_path)\n\n# # 转换为 TFLite 格式\n# converter = tf.lite.TFLiteConverter.from_saved_model(saved_model_path)\n# tflite_model = converter.convert()\n\n# # 保存 TFLite 模型\n# with open('/kaggle/working/model.tflite', 'wb') as f:\n#     f.write(tflite_model)\n\n# # 压缩模型文件\n# !zip submission.zip /kaggle/working/model.tflite\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.799834Z","iopub.status.idle":"2024-12-05T23:24:59.800165Z","shell.execute_reply.started":"2024-12-05T23:24:59.800012Z","shell.execute_reply":"2024-12-05T23:24:59.800029Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Verify TFLite model can be loaded and used for prediction\n# !pip install tflite-runtime\n# import tflite_runtime.interpreter as tflite\n\n# interpreter = tflite.Interpreter(\"/kaggle/working/model.tflite\")\n# found_signatures = list(interpreter.get_signature_list().keys())\n# prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\n# output = prediction_fn(inputs=demo_raw_data)\n# sign = output['outputs'].argmax()\n\n# print(\"PRED : \", ORD2SIGN.get(sign), f'[{sign}]')\n# print(\"TRUE : \", train.sign.values[0], f'[{train.sign_ord.values[0]}]')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T23:24:59.801495Z","iopub.status.idle":"2024-12-05T23:24:59.801816Z","shell.execute_reply.started":"2024-12-05T23:24:59.801634Z","shell.execute_reply":"2024-12-05T23:24:59.801654Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}