{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314},{"sourceType":"datasetVersion","sourceId":5017706,"datasetId":2896564,"databundleVersionId":5087746},{"sourceType":"datasetVersion","sourceId":5315518,"datasetId":3036481,"databundleVersionId":5388737}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt \nimport matplotlib as mpl\nimport seaborn as sns\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob, sys, os, math, gc, sklearn, scipy \n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}') ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:56:35.502570Z","iopub.execute_input":"2026-03-07T08:56:35.503325Z","iopub.status.idle":"2026-03-07T08:57:02.173799Z","shell.execute_reply.started":"2026-03-07T08:56:35.503294Z","shell.execute_reply":"2026-03-07T08:57:02.173159Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24 ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.175022Z","iopub.execute_input":"2026-03-07T08:57:02.175431Z","iopub.status.idle":"2026-03-07T08:57:02.180098Z","shell.execute_reply.started":"2026-03-07T08:57:02.175407Z","shell.execute_reply":"2026-03-07T08:57:02.179322Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set to True because the preprocessing logic was changed (added motion features) \nPREPROCESS_DATA = True \nTRAIN_MODEL = True\nUSE_VAL = True  # Keep True for tuning, set False for final Kaggle submission\n\n# Mixed Precision on P100 acts as a turbo boost: 2x faster training, 50% less VRAM usage\nfrom tensorflow.keras import mixed_precision\npolicy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\n\n# Data Dimensions\nN_ROWS = 543\nN_DIMS = 3\nDIM_NAMES = ['x', 'y', 'z']\nSEED = 42\nNUM_CLASSES = 250\nIS_INTERACTIVE = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '') == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\n# INPUT CONFIGURATION\n# Increased to 128 to capture fine temporal details (fingerspelling)\nINPUT_SIZE = 128 \n\n# Training Hyperparameters\nBATCH_ALL_SIGNS_N = 4\nBATCH_SIZE = 64\nN_EPOCHS = 200 # 100\nLR_MAX =  4e-4 #1e-3\nN_WARMUP_EPOCHS = 10 # Slight warmup helps Transformer stability\nWD_RATIO = 0.05\nMASK_VAL = 0.0 \n\n# Data type config\nDATA_DTYPE = np.float16\n\nprint(f'Compute Policy: {policy.compute_dtype}')\nprint(f'Variable Policy: {policy.variable_dtype}')","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.180948Z","iopub.execute_input":"2026-03-07T08:57:02.181239Z","iopub.status.idle":"2026-03-07T08:57:02.205268Z","shell.execute_reply.started":"2026-03-07T08:57:02.181212Z","shell.execute_reply":"2026-03-07T08:57:02.204660Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}') ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.206870Z","iopub.execute_input":"2026-03-07T08:57:02.207161Z","iopub.status.idle":"2026-03-07T08:57:02.214731Z","shell.execute_reply.started":"2026-03-07T08:57:02.207135Z","shell.execute_reply":"2026-03-07T08:57:02.214178Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read Training Data\n#if IS_INTERACTIVE or not PREPROCESS_DATA:\n#train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\n#else:\ntrain = pd.read_csv('/kaggle/input/competitions/asl-signs/train.csv')\n\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}') ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.215378Z","iopub.execute_input":"2026-03-07T08:57:02.215627Z","iopub.status.idle":"2026-03-07T08:57:02.424845Z","shell.execute_reply.started":"2026-03-07T08:57:02.215601Z","shell.execute_reply":"2026-03-07T08:57:02.424083Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    #return f'/kaggle/input/asl-signs/{path}'\n    return f'/kaggle/input/competitions/asl-signs/{path}'\n    \n\ntrain['file_path'] = train['path'].apply(get_file_path) #","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.425888Z","iopub.execute_input":"2026-03-07T08:57:02.426207Z","iopub.status.idle":"2026-03-07T08:57:02.460967Z","shell.execute_reply.started":"2026-03-07T08:57:02.426170Z","shell.execute_reply":"2026-03-07T08:57:02.460348Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ordinally Encode Sign","metadata":{}},{"cell_type":"code","source":"# Add ordinally Encoded Sign (assign number to each sign name)\ntrain['sign_ord'] = train['sign'].astype('category').cat.codes\n\n# Dictionaries to translate sign to ordinal encoded sign\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict() ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.461951Z","iopub.execute_input":"2026-03-07T08:57:02.462273Z","iopub.status.idle":"2026-03-07T08:57:02.558667Z","shell.execute_reply.started":"2026-03-07T08:57:02.462244Z","shell.execute_reply":"2026-03-07T08:57:02.557985Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train.sample(n=5)) \ndisplay(train.info()) ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.559502Z","iopub.execute_input":"2026-03-07T08:57:02.559799Z","iopub.status.idle":"2026-03-07T08:57:02.603153Z","shell.execute_reply.started":"2026-03-07T08:57:02.559777Z","shell.execute_reply":"2026-03-07T08:57:02.602583Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"N = int(1e3) if (IS_INTERACTIVE or not PREPROCESS_DATA) else int(10e3)\nN_UNIQUE_FRAMES = np.zeros(N, dtype=np.uint16)\nN_MISSING_FRAMES = np.zeros(N, dtype=np.uint16)\nMAX_FRAME = np.zeros(N, dtype=np.uint16)\n\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999]\n\nfor idx, file_path in enumerate(tqdm(train['file_path'].sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    N_UNIQUE_FRAMES[idx] = df['frame'].nunique()\n    N_MISSING_FRAMES[idx] = (df['frame'].max() - df['frame'].min()) - df['frame'].nunique() + 1\n    MAX_FRAME[idx] = df['frame'].max() ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:02.604009Z","iopub.execute_input":"2026-03-07T08:57:02.604304Z","iopub.status.idle":"2026-03-07T08:57:30.303461Z","shell.execute_reply.started":"2026-03-07T08:57:02.604274Z","shell.execute_reply":"2026-03-07T08:57:30.302742Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_UNIQUE_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+25, 25))\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:30.305678Z","iopub.execute_input":"2026-03-07T08:57:30.305946Z","iopub.status.idle":"2026-03-07T08:57:30.676232Z","shell.execute_reply.started":"2026-03-07T08:57:30.305924Z","shell.execute_reply":"2026-03-07T08:57:30.675666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of missing frames, consecutive frames with missing intermediate frame, i.e. 1,2,4,5 but 3 is missing\ndisplay(pd.Series(N_MISSING_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_MISSING_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Missing Frames', size=24)\npd.Series(N_MISSING_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:30.676986Z","iopub.execute_input":"2026-03-07T08:57:30.677233Z","iopub.status.idle":"2026-03-07T08:57:30.917234Z","shell.execute_reply.started":"2026-03-07T08:57:30.677209Z","shell.execute_reply":"2026-03-07T08:57:30.916712Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid() \nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:30.917972Z","iopub.execute_input":"2026-03-07T08:57:30.918207Z","iopub.status.idle":"2026-03-07T08:57:31.168215Z","shell.execute_reply.started":"2026-03-07T08:57:30.918179Z","shell.execute_reply":"2026-03-07T08:57:31.167475Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Landmark Indices for Lips, Hands, and Pose","metadata":{}},{"cell_type":"code","source":"USE_TYPES = ['left_hand', 'pose', 'right_hand']\nSTART_IDX = 468\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n# Landmark indices in original data\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n# Landmark indices in processed data\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\nprint(f'# HAND_IDXS: {len(HAND_IDXS)}, N_COLS: {N_COLS}')","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:31.169273Z","iopub.execute_input":"2026-03-07T08:57:31.169506Z","iopub.status.idle":"2026-03-07T08:57:31.177545Z","shell.execute_reply.started":"2026-03-07T08:57:31.169485Z","shell.execute_reply":"2026-03-07T08:57:31.176857Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}') ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:31.178458Z","iopub.execute_input":"2026-03-07T08:57:31.178739Z","iopub.status.idle":"2026-03-07T08:57:31.191680Z","shell.execute_reply.started":"2026-03-07T08:57:31.178713Z","shell.execute_reply":"2026-03-07T08:57:31.191183Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Process Data Tensorflow","metadata":{}},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32) ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:31.192509Z","iopub.execute_input":"2026-03-07T08:57:31.192800Z","iopub.status.idle":"2026-03-07T08:57:31.202878Z","shell.execute_reply.started":"2026-03-07T08:57:31.192779Z","shell.execute_reply":"2026-03-07T08:57:31.202194Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Removed strict input_signature to support Mixed Precision (Float16)\nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__() \n        self.lips_idxs = LIPS_IDXS\n        self.left_hand_idxs = LEFT_HAND_IDXS\n        self.pose_idxs = POSE_IDXS\n        self.landmark_idxs_left = LANDMARK_IDXS_LEFT_DOMINANT0\n        self.landmark_idxs_right = LANDMARK_IDXS_RIGHT_DOMINANT0\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function\n    def call(self, data0):\n        # CAST TO FLOAT32: Ensure numeric stability for Normalization\n        # Even if input is float16 (from Mixed Precision), we normalize in float32\n        data0 = tf.cast(data0, tf.float32)\n\n        # Filter dominant hand \n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0.0, 1.0))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0.0, 1.0))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Filter frames where the dominant hand is present\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0.0, 1.0), axis=[1, 2]\n            )\n            # Select relevant landmarks (Lips + Left Hand + Pose)\n            data = tf.gather(data0, self.landmark_idxs_left, axis=1)\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0.0, 1.0), axis=[1, 2]\n            )\n            # Select relevant landmarks (Lips + Right Hand + Pose)\n            data = tf.gather(data0, self.landmark_idxs_right, axis=1)\n            # Mirror X coordinate for right-handers to make them look left-handed\n            data = tf.concat([\n                -1.0 * tf.expand_dims(data[:, :, 0], axis=-1), # Flip X\n                tf.expand_dims(data[:, :, 1], axis=-1),        # Keep Y\n                tf.expand_dims(data[:, :, 2], axis=-1)         # Keep Z\n            ], axis=-1)\n            \n        # Get indices of valid frames\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        \n        # Filter the data to only keep valid frames\n        data = tf.gather(data, non_empty_frames_idxs, axis=0)\n\n        # Grandmaster Normalization (Anchor to Lips/Nose)\n        # We calculate the mean of the lips to use as the center (0,0,0)\n        lips = data[:, :40, :] \n        \n        # Calculate robust mean excluding NaNs\n        lips_mean = tf.math.reduce_mean(tf.where(tf.math.is_nan(lips), 0.0, lips), axis=1, keepdims=True)\n        lips_std = tf.math.reduce_std(tf.where(tf.math.is_nan(data), 0.0, data), axis=[1,2], keepdims=True) + 1e-6\n        \n        # Normalize: Center on lips, scale by global std\n        data = (data - lips_mean) / lips_std\n        \n        # Fill NaNs with 0.0 after normalization\n        data = tf.where(tf.math.is_nan(data), 0.0, data)\n\n        # Resizing / Interpolation to INPUT_SIZE (128)\n        N_FRAMES = tf.shape(data)[0]\n        \n        if N_FRAMES < INPUT_SIZE:\n            # Pad with zeros at the end\n            non_empty_frames_idxs = tf.pad(\n                tf.cast(non_empty_frames_idxs, tf.float32), \n                [[0, INPUT_SIZE - N_FRAMES]], \n                constant_values=-1\n            )\n            data = tf.pad(data, [[0, INPUT_SIZE - N_FRAMES], [0,0], [0,0]], constant_values=0)\n            \n        else:\n            # Downsample using Nearest Neighbor / Interpolation\n            data_flat = tf.reshape(data, [1, N_FRAMES, -1, 1])\n            data_resized = tf.image.resize(\n                data_flat, \n                [INPUT_SIZE, tf.shape(data_flat)[2]], \n                method=tf.image.ResizeMethod.BILINEAR\n            )\n            # Reshape back to (INPUT_SIZE, Points, Dims)\n            data = tf.reshape(data_resized, [INPUT_SIZE, -1, N_DIMS])\n            \n            # For indices, we take a linear spacing\n            non_empty_frames_idxs = tf.linspace(0.0, tf.cast(N_FRAMES, tf.float32), INPUT_SIZE)\n\n        # Motion Features (Lag1 & Lag2) \n        # Velocity (dX): x[t] - x[t-1]\n        dx = data[1:, :, :] - data[:-1, :, :]\n        dx = tf.concat([tf.zeros_like(data[:1, :, :]), dx], axis=0) # Pad start\n        \n        # Acceleration (ddX): dx[t] - dx[t-1]\n        ddx = dx[1:, :, :] - dx[:-1, :, :]\n        ddx = tf.concat([tf.zeros_like(dx[:1, :, :]), ddx], axis=0) # Pad start\n        \n        # Concatenate features: [X, Y, Z, dX, dY, dZ, ddX, ddY, ddZ]\n        data = tf.concat([data, dx, ddx], axis=-1)\n        \n        # Flatten points and features for the dense input layer\n        data = tf.reshape(data, (INPUT_SIZE, -1))\n        \n        return data, non_empty_frames_idxs\n\npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:31.203795Z","iopub.execute_input":"2026-03-07T08:57:31.204651Z","iopub.status.idle":"2026-03-07T08:57:31.221789Z","shell.execute_reply.started":"2026-03-07T08:57:31.204621Z","shell.execute_reply":"2026-03-07T08:57:31.221112Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n        \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data ","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:31.222719Z","iopub.execute_input":"2026-03-07T08:57:31.223003Z","iopub.status.idle":"2026-03-07T08:57:31.236670Z","shell.execute_reply.started":"2026-03-07T08:57:31.222974Z","shell.execute_reply":"2026-03-07T08:57:31.236084Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Memory Optimized: Split Metadata FIRST to avoid \"Copy Spike\" OOM\n\ndef preprocess_data():\n    # Infer Feature Shape\n    print(\"Determining feature shape...\")\n    dummy_path = train['file_path'].values[0]\n    dummy_data, _ = get_data(dummy_path)\n    N_COLS_FINAL = dummy_data.shape[1]\n    print(f\"Features per frame: {N_COLS_FINAL}\")\n\n    # Split Metadata first\n    print(\"Splitting metadata...\")\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\n    PARTICIPANT_IDS = train['participant_id'].values\n    train_idxs, val_idxs = next(splitter.split(train, train['sign_ord'], groups=PARTICIPANT_IDS))\n    \n    # Create DataFrames for Train and Val\n    df_train = train.iloc[train_idxs]\n    df_val = train.iloc[val_idxs]\n    \n    print(f\"Train Samples: {len(df_train)}\")\n    print(f\"Val Samples: {len(df_val)}\")\n\n    # Global Variable Setup\n    global X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN\n    global X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL, validation_data\n\n    # Process Training data\n    print(f\"Allocating X_train (approx {len(df_train)*INPUT_SIZE*N_COLS_FINAL*2 / 1e9:.2f} GB)...\")\n    \n    # Allocate Train Arrays\n    X_train = np.zeros([len(df_train), INPUT_SIZE, N_COLS_FINAL], dtype=np.float16)\n    y_train = np.zeros([len(df_train)], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.full([len(df_train), INPUT_SIZE], -1, dtype=np.float16)\n    \n    print(\"Filling X_train...\")\n    # Iterate over Train Metadata only\n    for i, (file_path, sign_ord) in enumerate(tqdm(df_train[['file_path', 'sign_ord']].values)):\n        if i % 5000 == 0: gc.collect()\n        \n        data, non_empty_frame_idxs = get_data(file_path)\n        \n        X_train[i] = data.numpy().astype(np.float16)\n        y_train[i] = sign_ord\n        NON_EMPTY_FRAME_IDXS_TRAIN[i] = non_empty_frame_idxs.numpy().astype(np.float16)\n        \n        if np.isnan(data).sum() > 0:\n            X_train[i] = np.nan_to_num(X_train[i])\n\n    # Process Validation Data (If enabled)\n    if USE_VAL:\n        print(\"Processing Validation Data...\")\n        X_val = np.zeros([len(df_val), INPUT_SIZE, N_COLS_FINAL], dtype=np.float16)\n        y_val = np.zeros([len(df_val)], dtype=np.int32)\n        NON_EMPTY_FRAME_IDXS_VAL = np.full([len(df_val), INPUT_SIZE], -1, dtype=np.float16)\n        \n        for i, (file_path, sign_ord) in enumerate(tqdm(df_val[['file_path', 'sign_ord']].values)):\n            data, non_empty_frame_idxs = get_data(file_path)\n            \n            X_val[i] = data.numpy().astype(np.float16)\n            y_val[i] = sign_ord\n            NON_EMPTY_FRAME_IDXS_VAL[i] = non_empty_frame_idxs.numpy().astype(np.float16)\n\n        # Create Validation Tuple\n        y_val_oh = tf.one_hot(y_val, NUM_CLASSES)\n        validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val_oh)\n    else:\n        validation_data = None\n\n    return N_COLS_FINAL\n\n# Execute\nif PREPROCESS_DATA:\n    N_COLS_FINAL = preprocess_data()","metadata":{"execution":{"iopub.status.busy":"2026-03-07T08:57:31.237368Z","iopub.execute_input":"2026-03-07T08:57:31.237582Z","iopub.status.idle":"2026-03-07T09:33:29.352612Z","shell.execute_reply.started":"2026-03-07T08:57:31.237562Z","shell.execute_reply":"2026-03-07T09:33:29.352067Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data Verification: Checks the in-memory arrays\n\n# Verify N_COLS_FINAL is set\nif 'N_COLS_FINAL' not in locals():\n    # Fallback inference\n    N_COLS_FINAL = X_train.shape[2]\n\nprint(\"-\" * 30) \nprint(f\"Features per frame (N_COLS_FINAL): {N_COLS_FINAL}\")\nprint(f\"X_train shape: {X_train.shape}  | dtype: {X_train.dtype}\")\nprint(f\"y_train shape: {y_train.shape}  | dtype: {y_train.dtype}\")\nprint(\"-\" * 30)\n\nif USE_VAL:\n    print(f\"X_val shape:   {X_val.shape}    | dtype: {X_val.dtype}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:29.353499Z","iopub.execute_input":"2026-03-07T09:33:29.353794Z","iopub.status.idle":"2026-03-07T09:33:29.361945Z","shell.execute_reply.started":"2026-03-07T09:33:29.353770Z","shell.execute_reply":"2026-03-07T09:33:29.361413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class Count\ndisplay(pd.Series(y_train).value_counts().to_frame('Class Count').iloc[[0,1,2,3,4, -5,-4,-3,-2,-1]]) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:29.362626Z","iopub.execute_input":"2026-03-07T09:33:29.362889Z","iopub.status.idle":"2026-03-07T09:33:29.423208Z","shell.execute_reply.started":"2026-03-07T09:33:29.362869Z","shell.execute_reply":"2026-03-07T09:33:29.422564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Vast majority of samples fits has less than 32 non empty frames\nN_EMPTY_FRAMES = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum(axis=1) \nN_EMPTY_FRAMES_WATERFALL = []\nfor n in tqdm(range(1,INPUT_SIZE+1)):\n    N_EMPTY_FRAMES_WATERFALL.append(sum(N_EMPTY_FRAMES >= n) / len(NON_EMPTY_FRAME_IDXS_TRAIN) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Non Empty Frames')\npd.Series(N_EMPTY_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks(np.arange(INPUT_SIZE), np.arange(1, INPUT_SIZE+1))\nplt.xlabel('Number of Non Empty Frames', size=16)\nplt.yticks(np.arange(0, 100+10, 10))\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Non Empty Frames', size=16)\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:29.424346Z","iopub.execute_input":"2026-03-07T09:33:29.424625Z","iopub.status.idle":"2026-03-07T09:33:31.284034Z","shell.execute_reply.started":"2026-03-07T09:33:29.424604Z","shell.execute_reply":"2026-03-07T09:33:31.283314Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Samples","metadata":{}},{"cell_type":"code","source":"# Features: Standard Batching + MixUp + Feature Masking + Low FPS Simulation\n\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, batch_size=32, mixup_alpha=0.2):\n    \"\"\"\n    Yields batches of 'batch_size' with robust On-the-Fly Augmentation.\n    Includes 'Low FPS' simulation to teach the model to handle laggy video.\n    \"\"\"\n    n_samples = len(X)\n    indices = np.arange(n_samples)\n    \n    while True:\n        # Shuffle indices at the start of every epoch\n        np.random.shuffle(indices)\n        \n        # Iterate through the dataset in chunks\n        for start_idx in range(0, n_samples, batch_size):\n            end_idx = min(start_idx + batch_size, n_samples)\n            \n            # Skip incomplete last batch if it's too small (avoids stability issues)\n            if end_idx - start_idx < 4: \n                continue\n                \n            batch_idxs = indices[start_idx:end_idx]\n            current_batch_size = len(batch_idxs)\n            \n            # Shape: (Batch, 128, Features)\n            X_batch = X[batch_idxs].astype(np.float32) # Cast to float32 for MixUp math\n            y_batch = tf.one_hot(y[batch_idxs], NUM_CLASSES).numpy()\n            non_empty_frame_idxs_batch = NON_EMPTY_FRAME_IDXS[batch_idxs]\n\n            # Augmentations (On-the-fly)\n            for b in range(current_batch_size):\n                \n                # Feature Masking (Dropout on inputs)\n                # Randomly zero out 20% of the features to force robustness\n                if np.random.rand() < 0.2:\n                    mask_indices = np.random.choice(N_COLS_FINAL, int(N_COLS_FINAL*0.2), replace=False)\n                    X_batch[b, :, mask_indices] = 0.0\n\n                # Time Masking (Drop contiguous frames)\n                # Simulates packet loss / dropped connection\n                if np.random.rand() < 0.3:\n                    len_mask = np.random.randint(5, 15)\n                    start_mask = np.random.randint(0, INPUT_SIZE - len_mask)\n                    X_batch[b, start_mask:start_mask+len_mask, :] = 0.0\n                    non_empty_frame_idxs_batch[b, start_mask:start_mask+len_mask] = -1\n                \n                # Low FPS Simulation (Random Frame Skipping)\n                # Drops every 2nd or 3rd frame to simulate 15fps/10fps webcam input\n                if np.random.rand() < 0.3:\n                    step = np.random.randint(2, 4) # Keep every 2nd or 3rd frame\n                    # Create indices [0, 2, 4...]\n                    frame_indices = np.arange(0, INPUT_SIZE, step)\n                    \n                    # Gather kept frames\n                    kept_frames = X_batch[b, frame_indices, :]\n                    kept_masks = non_empty_frame_idxs_batch[b, frame_indices]\n                    \n                    # Zero out the whole slot\n                    X_batch[b] = 0.0\n                    non_empty_frame_idxs_batch[b] = -1\n                    \n                    # Place kept frames at the start (compressing the sequence)\n                    # This teaches the model that \"fast\" sequences are valid\n                    valid_len = len(kept_frames)\n                    X_batch[b, :valid_len, :] = kept_frames\n                    non_empty_frame_idxs_batch[b, :valid_len] = kept_masks\n\n            # 3. MixUp (Blend 2 samples)\n            if mixup_alpha > 0 and np.random.rand() < 0.5:\n                lam = np.random.beta(mixup_alpha, mixup_alpha)\n                # Permute within the current batch\n                perm_indices = np.random.permutation(current_batch_size)\n                \n                X_batch = lam * X_batch + (1 - lam) * X_batch[perm_indices]\n                y_batch = lam * y_batch + (1 - lam) * y_batch[perm_indices]\n                \n                # For masks, take the dominant one (lambda > 0.5)\n                if lam < 0.5:\n                    non_empty_frame_idxs_batch = non_empty_frame_idxs_batch[perm_indices]\n            \n            yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:31.285079Z","iopub.execute_input":"2026-03-07T09:33:31.285318Z","iopub.status.idle":"2026-03-07T09:33:31.295421Z","shell.execute_reply.started":"2026-03-07T09:33:31.285296Z","shell.execute_reply":"2026-03-07T09:33:31.294832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN)\nX_batch, y_batch = next(dummy_dataset)\n\nfor k, v in X_batch.items():\n    print(f'{k} shape: {v.shape}, dtype: {v.dtype}')\n\n# Batch shape/dtype\nprint(f'y_batch shape: {y_batch.shape}, dtype: {y_batch.dtype}')\n# Verify each batch contains each sign exactly N times\ndisplay(pd.Series(np.argmax(y_batch, axis=1)).value_counts().to_frame('Counts')) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:31.296233Z","iopub.execute_input":"2026-03-07T09:33:31.296465Z","iopub.status.idle":"2026-03-07T09:33:31.356446Z","shell.execute_reply.started":"2026-03-07T09:33:31.296436Z","shell.execute_reply":"2026-03-07T09:33:31.355671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Config","metadata":{}},{"cell_type":"code","source":"# Config\nembed_dim = 192\nnum_heads = 4\nff_dim = embed_dim * 2\ndropout_rate = 0.2\n\n# Efficient Channel Attention (ECA)\nclass EcaLayer(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=5, **kwargs):\n        super().__init__(**kwargs)\n        self.kernel_size = kernel_size\n        self.conv = tf.keras.layers.Conv1D(1, kernel_size=kernel_size, padding='same', use_bias=False)\n\n    def call(self, x):\n        attn = tf.reduce_mean(x, axis=1, keepdims=True)\n        attn = tf.transpose(attn, (0, 2, 1))\n        attn = self.conv(attn)\n        attn = tf.transpose(attn, (0, 2, 1))\n        attn = tf.math.sigmoid(attn)\n        return x * attn\n\n# Conv1D Block\nclass Conv1DBlock(tf.keras.layers.Layer):\n    def __init__(self, dim, kernel_size=11, drop_rate=0.2, expand=4):\n        super().__init__()\n        self.conv = tf.keras.layers.DepthwiseConv1D(kernel_size, padding='same', use_bias=False)\n        self.bn = tf.keras.layers.BatchNormalization()\n        self.act = tf.keras.layers.Activation('swish') # Compatible activation\n        self.se = EcaLayer(kernel_size=5)\n        self.project = tf.keras.layers.Dense(dim, use_bias=False)\n        self.drop = tf.keras.layers.Dropout(drop_rate)\n        \n    def call(self, x, training=None): # Make training optional\n        skip = x\n        x = self.conv(x)\n        x = self.bn(x, training=training)\n        x = self.act(x)\n        x = self.se(x)\n        x = self.project(x)\n        if training:\n            x = self.drop(x)\n        return x + skip\n\n# Transformer Block\nclass TransformerBlock(tf.keras.layers.Layer):\n    def __init__(self, embed_dim, num_heads, ff_dim, rate=0.1):\n        super(TransformerBlock, self).__init__()\n        self.att = tf.keras.layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = tf.keras.Sequential([\n            tf.keras.layers.Dense(ff_dim, activation=\"gelu\"),\n            tf.keras.layers.Dense(embed_dim),\n        ])\n        self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = tf.keras.layers.Dropout(rate)\n        self.dropout2 = tf.keras.layers.Dropout(rate)\n\n    # Set training=None by default\n    def call(self, inputs, training=None):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        \n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output) \n\nclass LearnablePositionalEmbedding(tf.keras.layers.Layer):\n    \"\"\"\n    Adds a learnable vector to each frame position so the Transformer \n    knows 'where' it is in the sequence.\n    \"\"\"\n    def __init__(self, max_len, embed_dim):\n        super().__init__()\n        self.pos_embedding = tf.keras.layers.Embedding(input_dim=max_len, output_dim=embed_dim)\n\n    def call(self, x):\n        # x shape: (Batch, Time, Features)\n        max_len = tf.shape(x)[1]\n        positions = tf.range(start=0, limit=max_len, delta=1)\n        return x + self.pos_embedding(positions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:31.357426Z","iopub.execute_input":"2026-03-07T09:33:31.357719Z","iopub.status.idle":"2026-03-07T09:33:31.370069Z","shell.execute_reply.started":"2026-03-07T09:33:31.357662Z","shell.execute_reply":"2026-03-07T09:33:31.369392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hybrid CNN-Transformer\n# Added MaskingLayer\n# Inserted LearnablePositionalEmbedding\n\nclass MaskingLayer(tf.keras.layers.Layer):\n    def __init__(self, **kwargs):\n        super(MaskingLayer, self).__init__(**kwargs)\n    \n    def call(self, inputs):\n        frames, non_empty_frame_idxs = inputs\n        mask = tf.math.not_equal(non_empty_frame_idxs, -1)\n        mask = tf.cast(mask, dtype=frames.dtype)\n        mask_expanded = tf.expand_dims(mask, -1)\n        return frames * mask_expanded\n\ndef get_model():\n    # Input Shapes\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS_FINAL], dtype=tf.float16, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float16, name='non_empty_frame_idxs')\n    \n    # Apply Masking\n    x = MaskingLayer(name='input_masking')([frames, non_empty_frame_idxs])\n    \n    # Stem\n    x = tf.keras.layers.Dense(embed_dim, use_bias=False, name='stem_conv')(x)\n    x = tf.keras.layers.BatchNormalization(momentum=0.95, name='stem_bn')(x)\n    \n    # Backbone\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    \n    x = LearnablePositionalEmbedding(INPUT_SIZE, embed_dim)(x)\n    \n    # Transformer\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, rate=0.2)(x)\n    \n    # CNNs\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    \n    # Transformer\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, rate=0.2)(x)\n    \n    #Head\n    x = tf.keras.layers.GlobalAveragePooling1D()(x)\n    x = tf.keras.layers.Dropout(0.8)(x) \n    outputs = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax', dtype='float32', name='classifier')(x)\n    \n    # Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Optimizer (Native Weight Decay)\n    optimizer = tf.optimizers.AdamW(learning_rate=LR_MAX, weight_decay=WD_RATIO)\n    loss = tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1)\n    \n    metrics = [\n        tf.keras.metrics.CategoricalAccuracy(name='acc'),\n        tf.keras.metrics.TopKCategoricalAccuracy(k=5, name='top_5_acc'),\n    ]\n    \n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model\n\ntf.keras.backend.clear_session()\nmodel = get_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:31.371024Z","iopub.execute_input":"2026-03-07T09:33:31.371384Z","iopub.status.idle":"2026-03-07T09:33:34.666866Z","shell.execute_reply.started":"2026-03-07T09:33:31.371356Z","shell.execute_reply":"2026-03-07T09:33:34.666289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:34.667618Z","iopub.execute_input":"2026-03-07T09:33:34.667896Z","iopub.status.idle":"2026-03-07T09:33:35.524263Z","shell.execute_reply.started":"2026-03-07T09:33:34.667874Z","shell.execute_reply":"2026-03-07T09:33:35.523470Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# No NaN Predictions","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    y_pred = model.predict_on_batch(X_batch).flatten()\n    print(f'# NaN Values In Prediction: {np.isnan(y_pred).sum()}') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:35.525153Z","iopub.execute_input":"2026-03-07T09:33:35.525393Z","iopub.status.idle":"2026-03-07T09:33:35.529273Z","shell.execute_reply.started":"2026-03-07T09:33:35.525363Z","shell.execute_reply":"2026-03-07T09:33:35.528593Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Weight Iitialization","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    plt.figure(figsize=(12,5))\n    plt.title(f'Softmax Output Initialized Model | µ={y_pred.mean():.3f}, σ={y_pred.std():.3f}', pad=25)\n    pd.Series(y_pred).plot(kind='hist', bins=128, label='Class Probability')\n    plt.xlim(0, max(y_pred) * 1.1)\n    plt.vlines([1 / NUM_CLASSES], 0, plt.ylim()[1], color='red', label=f'Random Guessing Baseline 1/NUM_CLASSES={1 / NUM_CLASSES:.3f}')\n    plt.grid()\n    plt.legend()\n    plt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:35.533076Z","iopub.execute_input":"2026-03-07T09:33:35.533585Z","iopub.status.idle":"2026-03-07T09:33:35.547759Z","shell.execute_reply.started":"2026-03-07T09:33:35.533562Z","shell.execute_reply":"2026-03-07T09:33:35.547054Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Learning Rate Scheduler","metadata":{}},{"cell_type":"code","source":"# Calculate total steps\nSTEPS_PER_EPOCH = N_SAMPLES // BATCH_SIZE\nTOTAL_STEPS = N_EPOCHS * STEPS_PER_EPOCH\n\ndef generate_one_cycle_step_wise(lr_max, total_steps, warmup_prop=0.3):\n    lr_schedule = []\n    warmup_steps = int(total_steps * warmup_prop)\n    cooldown_steps = total_steps - warmup_steps\n    \n    initial_lr = lr_max / 25.0\n    final_lr = lr_max / 10000.0\n    \n    for step in range(total_steps):\n        if step < warmup_steps:\n            lr = initial_lr + (lr_max - initial_lr) * (step / warmup_steps)\n        else:\n            progress = (step - warmup_steps) / cooldown_steps\n            lr = final_lr + (lr_max - final_lr) * (0.5 * (1 + math.cos(math.pi * progress)))\n        lr_schedule.append(lr)\n    return lr_schedule\n\n# Generate high-resolution schedule\nLR_SCHEDULE_STEPS = generate_one_cycle_step_wise(LR_MAX, TOTAL_STEPS, warmup_prop=0.3)\n\n# Custom Callback to update every BATCH\nclass StepLearningRateScheduler(tf.keras.callbacks.Callback):\n    def __init__(self, schedule):\n        super(StepLearningRateScheduler, self).__init__()\n        self.schedule = schedule\n\n    def on_train_batch_begin(self, batch, logs=None):\n        # Calculate global step\n        global_step = self.model.optimizer.iterations.numpy()\n        if global_step < len(self.schedule):\n            lr = self.schedule[global_step]\n            self.model.optimizer.learning_rate.assign(lr)\n\nlr_callback = StepLearningRateScheduler(LR_SCHEDULE_STEPS) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:35.548679Z","iopub.execute_input":"2026-03-07T09:33:35.548975Z","iopub.status.idle":"2026-03-07T09:33:35.617531Z","shell.execute_reply.started":"2026-03-07T09:33:35.548945Z","shell.execute_reply":"2026-03-07T09:33:35.617047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Weight Decay Callback","metadata":{}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:35.618257Z","iopub.execute_input":"2026-03-07T09:33:35.618432Z","iopub.status.idle":"2026-03-07T09:33:35.622892Z","shell.execute_reply.started":"2026-03-07T09:33:35.618416Z","shell.execute_reply":"2026-03-07T09:33:35.622187Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Verify val dataset covers all signs\n    print(f'# Unique Signs in Validation Set: {pd.Series(y_val).nunique()}')\n    # Value Counts\n    display(pd.Series(y_val).value_counts().to_frame('Count').iloc[[1,2,3,-3,-2,-1]]) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:35.623588Z","iopub.execute_input":"2026-03-07T09:33:35.623858Z","iopub.status.idle":"2026-03-07T09:33:35.638529Z","shell.execute_reply.started":"2026-03-07T09:33:35.623829Z","shell.execute_reply":"2026-03-07T09:33:35.637841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sanity Check","metadata":{}},{"cell_type":"code","source":"# Sanity Check\nif TRAIN_MODEL and USE_VAL:\n    _ = model.evaluate(*validation_data, verbose=2) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:33:35.639912Z","iopub.execute_input":"2026-03-07T09:33:35.640115Z","iopub.status.idle":"2026-03-07T09:34:14.422251Z","shell.execute_reply.started":"2026-03-07T09:33:35.640097Z","shell.execute_reply":"2026-03-07T09:34:14.421594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL and USE_VAL:\n    try:\n        # validation_data is tuple (X, y)\n        if isinstance(validation_data, tuple) or isinstance(validation_data, list):\n            val_inputs, val_labels = validation_data\n            if len(val_labels.shape) == 1:\n                print('Converting validation labels to One-Hot')\n                val_labels_oh = tf.one_hot(val_labels, NUM_CLASSES)\n                validation_data = (val_inputs, val_labels_oh)\n    except Exception as e:\n        print(f'Warning: Could not auto-convert validation labels: {e}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:34:14.425787Z","iopub.execute_input":"2026-03-07T09:34:14.426035Z","iopub.status.idle":"2026-03-07T09:34:14.432579Z","shell.execute_reply.started":"2026-03-07T09:34:14.426014Z","shell.execute_reply":"2026-03-07T09:34:14.432121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Streams validation data in mini-batches to prevent GPU OOM at the end of the epoch\n\ndef get_val_batch(X, y, NON_EMPTY_FRAME_IDXS, batch_size=32):\n    n_samples = len(X)\n    indices = np.arange(n_samples)\n    \n    while True:\n        for start_idx in range(0, n_samples, batch_size):\n            end_idx = min(start_idx + batch_size, n_samples)\n            \n            batch_idxs = indices[start_idx:end_idx]\n            \n            # Fetch Data (Cast to float32 for model consistency)\n            X_batch = X[batch_idxs].astype(np.float32)\n            y_batch = tf.one_hot(y[batch_idxs], NUM_CLASSES).numpy()\n            non_empty_frame_idxs_batch = NON_EMPTY_FRAME_IDXS[batch_idxs]\n            \n            yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:34:14.433412Z","iopub.execute_input":"2026-03-07T09:34:14.433673Z","iopub.status.idle":"2026-03-07T09:34:14.453100Z","shell.execute_reply.started":"2026-03-07T09:34:14.433646Z","shell.execute_reply":"2026-03-07T09:34:14.452418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Config updates\nN_EPOCHS = 200            \nBATCH_SIZE = 32           \nWARMUP_EPOCHS = 20        \nINIT_LR = 1e-4\nMAX_LR = 4e-4             \nFINAL_LR = 1e-6\n\nSTEPS_PER_EPOCH = len(X_train) // BATCH_SIZE\n\nif USE_VAL:\n    VAL_STEPS = len(X_val) // BATCH_SIZE\nelse:\n    VAL_STEPS = None\n\nTOTAL_STEPS = N_EPOCHS * STEPS_PER_EPOCH\n\nprint(f\"   Epochs: {N_EPOCHS}\")\nprint(f\"   Train Steps: {STEPS_PER_EPOCH} | Val Steps: {VAL_STEPS}\")\n\n# Cosine Decay Schedule\nlr_schedule = tf.keras.optimizers.schedules.CosineDecay(\n    initial_learning_rate=MAX_LR,\n    decay_steps=TOTAL_STEPS,\n    alpha=FINAL_LR / MAX_LR\n)\n\n# Re-compile Model\noptimizer = tf.keras.optimizers.AdamW(learning_rate=lr_schedule, weight_decay=WD_RATIO)\nloss = tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1)\n\nmetrics = [\n    tf.keras.metrics.CategoricalAccuracy(name='acc'),\n    tf.keras.metrics.TopKCategoricalAccuracy(k=5, name='top_5_acc'),\n]\n\nmodel.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n\n# Fit Model\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', \n    patience=20,          \n    restore_best_weights=True, \n    verbose=1\n)\n\n\nif TRAIN_MODEL:\n    history = model.fit(\n            # Training Generator\n            x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN, batch_size=BATCH_SIZE),\n            steps_per_epoch=STEPS_PER_EPOCH,\n            \n            # Validation Generator (CRITICAL OOM FIX)\n            validation_data=get_val_batch(X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL, batch_size=BATCH_SIZE) if USE_VAL else None,\n            validation_steps=VAL_STEPS if USE_VAL else None,\n            \n            epochs=N_EPOCHS,\n            callbacks=[early_stopping],\n            verbose=VERBOSE,\n        )\n    \n    model.save_weights('hybrid_model.weights.h5')\n    \n    # ── Save Training History to CSV ──────────────────────────────────────────────\n    history_df = pd.DataFrame(history.history)\n    history_df.index += 1          # epoch numbers start at 1\n    history_df.index.name = 'epoch'\n    history_df.to_csv('training_history.csv')\n    print(f\"Training history saved → training_history.csv  ({len(history_df)} epochs logged)\")\n    display(history_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T09:34:14.454187Z","iopub.execute_input":"2026-03-07T09:34:14.454430Z","iopub.status.idle":"2026-03-07T12:10:59.572903Z","shell.execute_reply.started":"2026-03-07T09:34:14.454411Z","shell.execute_reply":"2026-03-07T12:10:59.572038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#if USE_VAL:\n    # val dataset predictions\n    #y_val_pred = model.predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2).argmax(axis=1)\n    # Label\n    #labels = [ORD2SIGN.get(i).replace(' ', '_') for i in range(NUM_CLASSES)] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:10:59.574074Z","iopub.execute_input":"2026-03-07T12:10:59.574339Z","iopub.status.idle":"2026-03-07T12:10:59.577708Z","shell.execute_reply.started":"2026-03-07T12:10:59.574315Z","shell.execute_reply":"2026-03-07T12:10:59.577136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport ctypes\nimport pickle\n\n# PHASE 1: SECURE THE TRAINING HISTORY\n# We save this FIRST. If the kernel crashes during inference, \n# you can simply load this file in a new notebook to generate the training accuracy and loss plots.\nif 'history' in globals():\n    with open('history_backup.pkl', 'wb') as f:\n        pickle.dump(history.history, f)\n    print(\"Training history saved to 'history_backup.pkl'.\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:10:59.578501Z","iopub.execute_input":"2026-03-07T12:10:59.578789Z","iopub.status.idle":"2026-03-07T12:10:59.597184Z","shell.execute_reply.started":"2026-03-07T12:10:59.578765Z","shell.execute_reply":"2026-03-07T12:10:59.596650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# AGGRESSIVE MEMORY FLUSH \n# 1. Delete Training Data\nif 'X_train' in globals():\n    del X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN\n    print(\"🧹 Python variables deleted.\")\n\n# 2. Force Garbage Collection\ngc.collect()\n\n# 3. Force Malloc Trim (The Secret Weapon)\n# This forces the C-level memory allocator to release freed memory back to the OS.\ntry:\n    ctypes.CDLL(\"libc.so.6\").malloc_trim(0)\n    print(\"📉 System memory explicitly trimmed.\")\nexcept Exception as e:\n    print(f\"⚠️ Could not trim malloc: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:10:59.599040Z","iopub.execute_input":"2026-03-07T12:10:59.599585Z","iopub.status.idle":"2026-03-07T12:11:00.309778Z","shell.execute_reply.started":"2026-03-07T12:10:59.599557Z","shell.execute_reply":"2026-03-07T12:11:00.309141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STREAMED INFERENCE (Low Memory)\nif USE_VAL:\n    print(f\"Starting Streamed Inference on {len(X_val)} samples...\")\n    \n    # Pre-allocate the result array (Int16 is tiny compared to Float32 probabilities)\n    y_val_pred = np.zeros(len(X_val), dtype=np.int16)\n    \n    # Manual Batching Loop\n    # We predict small chunks and immediately Argmax to save RAM\n    chunk_size = 1000 # Process 1000 samples at a time (RAM safe)\n    sub_batch_size = 32 # Feed GPU 32 at a time (VRAM safe)\n    \n    total_chunks = int(np.ceil(len(X_val) / chunk_size))\n    \n    for i in tqdm(range(total_chunks), desc=\"Streaming Predictions\"):\n        start = i * chunk_size\n        end = min((i + 1) * chunk_size, len(X_val))\n        \n        # 1. Select Slice\n        batch_frames = X_val[start:end]\n        batch_idxs = NON_EMPTY_FRAME_IDXS_VAL[start:end]\n        \n        # 2. Predict (Returns Probabilities)\n        # Using a fresh predict call for each chunk prevents memory accumulation\n        probs = model.predict(\n            {'frames': batch_frames, 'non_empty_frame_idxs': batch_idxs}, \n            batch_size=sub_batch_size, \n            verbose=0\n        )\n        \n        # 3. Argmax immediately (Compress 250 floats -> 1 int)\n        y_val_pred[start:end] = probs.argmax(axis=1).astype(np.int16)\n        \n        # 4. Explicit Cleanup\n        del probs, batch_frames, batch_idxs\n        gc.collect()\n    \n    # Generate Labels for Report\n    labels = [ORD2SIGN.get(i).replace(' ', '_') for i in range(NUM_CLASSES)]\n    print(\"Inference Complete.\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:11:00.310619Z","iopub.execute_input":"2026-03-07T12:11:00.310878Z","iopub.status.idle":"2026-03-07T12:11:43.651033Z","shell.execute_reply.started":"2026-03-07T12:11:00.310857Z","shell.execute_reply":"2026-03-07T12:11:43.650155Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Landmark Attention Weights","metadata":{}},{"cell_type":"code","source":"#embedding_layer = None\n#for layer in model.layers:\n #   if 'embedding' in layer.name:\n  #      embedding_layer = layer\n   #     break\n\n#if embedding_layer is None:\n #   raise RuntimeError(\"Embedding layer not found in the model.\")\n\n#landmark_weights_var = getattr(embedding_layer, 'landmark_weights', None)\n\n#if landmark_weights_var is None:\n #   raise RuntimeError(\"'landmark_weights' attribute not found in the embedding layer.\")\n\n# Convert to numpy and apply softmax for readable weights\n#import scipy\n\n#weights = scipy.special.softmax(landmark_weights_var.numpy())\n\n #Print weights\n#landmarks = ['lips_embedding', 'left_hand_embedding', 'pose_embedding']\n\n#for w, lm in zip(weights, landmarks):\n #   print(f'{lm} weight: {(w * 100):.1f}%') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:11:43.652098Z","iopub.execute_input":"2026-03-07T12:11:43.652347Z","iopub.status.idle":"2026-03-07T12:11:43.656493Z","shell.execute_reply.started":"2026-03-07T12:11:43.652324Z","shell.execute_reply":"2026-03-07T12:11:43.655748Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Classification Report","metadata":{}},{"cell_type":"code","source":"def print_classification_report():\n    # Classification report for all signs\n    classification_report = sklearn.metrics.classification_report(\n            y_val,\n            y_val_pred,\n            target_names=labels,\n            output_dict=True,\n        )\n    # Round Data for better readability\n    classification_report = pd.DataFrame(classification_report).T\n    classification_report = classification_report.round(2)\n    classification_report = classification_report.astype({\n            'support': np.uint16,\n        })\n    # Add signs\n    classification_report['sign'] = [e if e in SIGN2ORD else -1 for e in classification_report.index]\n    classification_report['sign_ord'] = classification_report['sign'].apply(SIGN2ORD.get).fillna(-1).astype(np.int16)\n    # Sort on F1-score\n    classification_report = pd.concat((\n        classification_report.head(NUM_CLASSES).sort_values('f1-score', ascending=False),\n        classification_report.tail(3),\n    ))\n\n    pd.options.display.max_rows = 999\n    display(classification_report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:11:43.657374Z","iopub.execute_input":"2026-03-07T12:11:43.657616Z","iopub.status.idle":"2026-03-07T12:11:43.673200Z","shell.execute_reply.started":"2026-03-07T12:11:43.657584Z","shell.execute_reply":"2026-03-07T12:11:43.672512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if USE_VAL:\n    print_classification_report() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:11:43.674071Z","iopub.execute_input":"2026-03-07T12:11:43.674412Z","iopub.status.idle":"2026-03-07T12:11:43.790761Z","shell.execute_reply.started":"2026-03-07T12:11:43.674381Z","shell.execute_reply":"2026-03-07T12:11:43.790042Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Report Analytics","metadata":{}},{"cell_type":"markdown","source":"## 1. Top-1 vs Top-5 Accuracy","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # ── Streamed Top-5 Inference ──────────────────────────────────────────────\n    # Re-run prediction to collect full probability distributions for Top-5\n    print('Computing Top-1 and Top-5 accuracy...')\n\n    top1_correct  = 0\n    top5_correct  = 0\n    total_samples = len(X_val)\n    chunk_size    = 1000\n    sub_batch     = 32\n\n    for i in range(int(np.ceil(total_samples / chunk_size))):\n        start = i * chunk_size\n        end   = min(start + chunk_size, total_samples)\n\n        probs = model.predict(\n            {'frames': X_val[start:end].astype(np.float32),\n             'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL[start:end]},\n            batch_size=sub_batch, verbose=0\n        )  # (chunk, 250)\n\n        true_labels   = y_val[start:end]\n        top1_preds    = probs.argmax(axis=1)\n        top5_preds    = np.argsort(probs, axis=1)[:, -5:]  # top-5 indices\n\n        top1_correct += (top1_preds == true_labels).sum()\n        top5_correct += np.array([true_labels[j] in top5_preds[j] for j in range(len(true_labels))]).sum()\n\n        del probs\n        import gc; gc.collect()\n\n    top1_acc = top1_correct / total_samples\n    top5_acc = top5_correct / total_samples\n\n    print(f'\\n{\"─\" * 40}')\n    print(f'  Top-1 Accuracy : {top1_acc:.4f}  ({top1_acc*100:.2f}%)')\n    print(f'  Top-5 Accuracy : {top5_acc:.4f}  ({top5_acc*100:.2f}%)')\n    print(f'  Gap (Top5-Top1): {(top5_acc - top1_acc)*100:.2f} pp')\n    print(f'{\"─\" * 40}\\n')\n\n    # ── Bar chart ──────────────────────────────────────────────────────────────\n    fig, ax = plt.subplots(figsize=(8, 5))\n    bars = ax.bar(['Top-1 Accuracy', 'Top-5 Accuracy'],\n                  [top1_acc * 100, top5_acc * 100],\n                  color=['#4C72B0', '#55A868'], width=0.45, edgecolor='black')\n    ax.set_ylim(0, 105)\n    ax.set_ylabel('Accuracy (%)', fontsize=14)\n    ax.set_title('Top-1 vs Top-5 Validation Accuracy (250 Classes)', fontsize=16, pad=12)\n    for bar, val in zip(bars, [top1_acc, top5_acc]):\n        ax.text(bar.get_x() + bar.get_width() / 2,\n                bar.get_height() + 1.5,\n                f'{val*100:.2f}%', ha='center', va='bottom', fontsize=13, fontweight='bold')\n    ax.axhline(100, color='red', linestyle='--', linewidth=0.8, alpha=0.5, label='Perfect Score')\n    ax.legend(fontsize=11)\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:11:43.791779Z","iopub.execute_input":"2026-03-07T12:11:43.792143Z","iopub.status.idle":"2026-03-07T12:12:14.016527Z","shell.execute_reply.started":"2026-03-07T12:11:43.792114Z","shell.execute_reply":"2026-03-07T12:12:14.015732Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Confusion Matrix – Top 20 Most Confused Sign Pairs","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    from sklearn.metrics import confusion_matrix\n\n    # y_val_pred already computed by the streamed inference cell above\n    cm = confusion_matrix(y_val, y_val_pred, labels=np.arange(NUM_CLASSES))  # (250, 250)\n\n    # Extract off-diagonal (confused) pairs \n    cm_no_diag = cm.copy()\n    np.fill_diagonal(cm_no_diag, 0)\n\n    # Flatten and find the top-10 confused (true_class, pred_class) pairs\n    flat_indices  = np.argsort(cm_no_diag.ravel())[::-1][:10]\n    top_pairs_rc  = np.unravel_index(flat_indices, cm_no_diag.shape)\n    top_true_idxs = top_pairs_rc[0]\n    top_pred_idxs = top_pairs_rc[1]\n\n    # Unique classes involved in the top-10 confused pairs\n    unique_classes = np.unique(np.concatenate([top_true_idxs, top_pred_idxs]))\n    unique_labels  = [ORD2SIGN[i] for i in unique_classes]\n\n    # Sub-matrix for those classes only\n    sub_cm = cm[np.ix_(unique_classes, unique_classes)]\n\n    # Plot heatmap \n    fig, ax = plt.subplots(figsize=(max(10, len(unique_classes)), max(8, len(unique_classes) * 0.7)))\n\n    # Normalise each row so we see confusion rate (not raw counts)\n    row_sums    = sub_cm.sum(axis=1, keepdims=True)\n    sub_cm_norm = np.where(row_sums > 0, sub_cm / row_sums, 0.0)\n\n    sns.heatmap(\n        sub_cm_norm,\n        annot=True, fmt='.2f',\n        xticklabels=unique_labels,\n        yticklabels=unique_labels,\n        cmap='YlOrRd',\n        linewidths=0.4,\n        ax=ax\n    )\n    ax.set_title(\n        'Confusion Matrix: Classes Involved in Top-10 Most Confused Pairs\\n'\n        '(Row-normalised – diagonal = correct; off-diagonal = confusion rate)',\n        fontsize=15, pad=14\n    )\n    ax.set_xlabel('Predicted Sign', fontsize=13)\n    ax.set_ylabel('True Sign',      fontsize=13)\n    ax.tick_params(axis='x', rotation=45)\n    ax.tick_params(axis='y', rotation=0)\n    plt.tight_layout()\n    # Save the figure as PNG\n    png_filename = 'confusion_matrix.png'\n    fig.savefig(png_filename, dpi=300, bbox_inches='tight')\n    \n    # Display the plot\n    plt.show()\n    print(f\"Confusion matrix saved and displayed as {png_filename}\")\n   \n\n    # Print top-10 confused pairs as a table \n    print('\\n Top 10 Most Confused Sign Pairs')\n    print(f'{\"Rank\":<4} {\"True Sign\":<12} {\"Predicted As\":<12} {\"Count\":>6}')\n    print('─' * 56)\n    for rank, (tr, pr) in enumerate(zip(top_true_idxs, top_pred_idxs), 1):\n        print(f'{rank:<4} {ORD2SIGN[tr]:<12} {ORD2SIGN[pr]:12} {cm_no_diag[tr, pr]:>6}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:12:14.017519Z","iopub.execute_input":"2026-03-07T12:12:14.017783Z","iopub.status.idle":"2026-03-07T12:12:16.588861Z","shell.execute_reply.started":"2026-03-07T12:12:14.017761Z","shell.execute_reply":"2026-03-07T12:12:16.588164Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Per-Class F1 Score – Best 5 & Worst 5 Signs","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    from sklearn.metrics import f1_score\n\n    # Per-class F1 (macro over each class individually)\n    per_class_f1 = f1_score(y_val, y_val_pred, labels=np.arange(NUM_CLASSES),\n                            average=None, zero_division=0)  # shape (250,)\n\n    # Build a tidy DataFrame\n    f1_df = pd.DataFrame({\n        'sign'     : [ORD2SIGN[i] for i in range(NUM_CLASSES)],\n        'f1_score' : per_class_f1,\n        'support'  : [(y_val == i).sum() for i in range(NUM_CLASSES)],\n    }).sort_values('f1_score', ascending=False).reset_index(drop=True)\n\n    best5  = f1_df.head(5)\n    worst5 = f1_df.tail(5).iloc[::-1]   # worst first\n\n    print('Best 5 Signs (highest F1)')\n    display(best5[['sign', 'f1_score', 'support']].reset_index(drop=True))\n\n    print('\\n Worst 5 Signs (lowest F1)')\n    display(worst5[['sign', 'f1_score', 'support']].reset_index(drop=True))\n\n    # Combined bar chart\n    combined = pd.concat([best5, worst5], ignore_index=True)\n    colors   = ['#2ecc71'] * 5 + ['#e74c3c'] * 5  # green / red\n\n    fig, ax = plt.subplots(figsize=(14, 6))\n    bars = ax.barh(\n        combined['sign'][::-1],\n        combined['f1_score'][::-1],\n        color=colors[::-1],\n        edgecolor='black', linewidth=0.6\n    )\n\n    # Annotate bars with F1 value and support\n    for bar, (_, row) in zip(bars, combined[::-1].iterrows()):\n        ax.text(\n            bar.get_width() + 0.005,\n            bar.get_y() + bar.get_height() / 2,\n            f'{row[\"f1_score\"]:.3f}  (n={row[\"support\"]})',\n            va='center', ha='left', fontsize=10\n        )\n\n    ax.set_xlim(0, 1.18)\n    ax.set_xlabel('F1 Score', fontsize=13)\n    ax.set_title('Per-Class F1 Score – Best 5 (green) vs Worst 5 (red) Signs', fontsize=15, pad=12)\n    ax.axvline(per_class_f1.mean(), color='navy', linestyle='--', linewidth=1.2,\n               label=f'Mean F1 = {per_class_f1.mean():.3f}')\n    ax.legend(fontsize=11)\n\n    # Divider line between best / worst\n    ax.axhline(4.5, color='black', linewidth=1.2, linestyle=':')\n    ax.text(0.01, 4.65, 'BEST 5', fontsize=9, color='#27ae60', fontweight='bold')\n    ax.text(0.01, 4.35, 'WORST 5', fontsize=9, color='#c0392b', fontweight='bold')\n\n    plt.tight_layout()\n    # Save the figure as PNG\n    png_filename = 'per_class_F1.png'\n    fig.savefig(png_filename, dpi=300, bbox_inches='tight')\n    \n    # Display the plot\n    plt.show()\n    \n    # Print confirmation\n    print(f\"Per Class F1 Score {png_filename}\")\n\n    # Overall summary \n    print(f'\\nOverall Macro-F1   : {per_class_f1.mean():.4f}')\n    print(f'Median Per-Class F1: {np.median(per_class_f1):.4f}')\n    print(f'Classes with F1 < 0.50: {(per_class_f1 < 0.50).sum()} / {NUM_CLASSES}')\n    print(f'Classes with F1 > 0.90: {(per_class_f1 > 0.90).sum()} / {NUM_CLASSES}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:12:16.589822Z","iopub.execute_input":"2026-03-07T12:12:16.590070Z","iopub.status.idle":"2026-03-07T12:12:17.339210Z","shell.execute_reply.started":"2026-03-07T12:12:16.590049Z","shell.execute_reply":"2026-03-07T12:12:17.338482Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training History","metadata":{}},{"cell_type":"code","source":"def plot_history_metric(metric, f_best=np.argmax, ylim=None, yscale=None, yticks=None):\n    plt.figure(figsize=(20, 10))\n    \n    values = history.history[metric]\n    N_EPOCHS = len(values)\n    val = 'val' in ''.join(history.history.keys())\n    # Epoch ticks\n    if N_EPOCHS <= 20:\n        x = np.arange(1, N_EPOCHS + 1)\n    else:\n        x = [1, 5] + [10 + 5 * idx for idx in range((N_EPOCHS - 10) // 5 + 1)]\n\n    x_ticks = np.arange(1, N_EPOCHS+1)\n\n    # Validation\n    if val:\n        val_values = history.history[f'val_{metric}']\n        val_argmin = f_best(val_values)\n        plt.plot(x_ticks, val_values, label=f'val')\n\n    # summarize history for accuracy\n    plt.plot(x_ticks, values, label=f'train')\n    argmin = f_best(values)\n    plt.scatter(argmin + 1, values[argmin], color='red', s=75, marker='o', label=f'train_best')\n    if val:\n        plt.scatter(val_argmin + 1, val_values[val_argmin], color='purple', s=75, marker='o', label=f'val_best')\n\n    plt.title(f'Model {metric}', fontsize=24, pad=10)\n    plt.ylabel(metric, fontsize=20, labelpad=10)\n\n    if ylim:\n        plt.ylim(ylim)\n\n    if yscale is not None:\n        plt.yscale(yscale)\n        \n    if yticks is not None:\n        plt.yticks(yticks, fontsize=16)\n\n    plt.xlabel('epoch', fontsize=20, labelpad=10)        \n    plt.tick_params(axis='x', labelsize=8)\n    plt.xticks(x, fontsize=16) # set tick step to 1 and let x axis start at 1\n    plt.yticks(fontsize=16)\n    \n    plt.legend(prop={'size': 10})\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:12:17.340267Z","iopub.execute_input":"2026-03-07T12:12:17.340628Z","iopub.status.idle":"2026-03-07T12:12:17.348627Z","shell.execute_reply.started":"2026-03-07T12:12:17.340605Z","shell.execute_reply":"2026-03-07T12:12:17.347932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('loss', f_best=np.argmin) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:12:17.349590Z","iopub.execute_input":"2026-03-07T12:12:17.349932Z","iopub.status.idle":"2026-03-07T12:12:17.655352Z","shell.execute_reply.started":"2026-03-07T12:12:17.349911Z","shell.execute_reply":"2026-03-07T12:12:17.654644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1)) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:12:17.656153Z","iopub.execute_input":"2026-03-07T12:12:17.656363Z","iopub.status.idle":"2026-03-07T12:12:17.926061Z","shell.execute_reply.started":"2026-03-07T12:12:17.656343Z","shell.execute_reply":"2026-03-07T12:12:17.925316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_5_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T12:12:17.927075Z","iopub.execute_input":"2026-03-07T12:12:17.927342Z","iopub.status.idle":"2026-03-07T12:12:18.203518Z","shell.execute_reply.started":"2026-03-07T12:12:17.927320Z","shell.execute_reply":"2026-03-07T12:12:18.202922Z"}},"outputs":[],"execution_count":null}]}