{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314},{"sourceType":"datasetVersion","sourceId":5017706,"datasetId":2896564,"databundleVersionId":5087746}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt \nimport matplotlib as mpl\nimport seaborn as sns\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob, sys, os, math, gc, sklearn, scipy, json \n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}') ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:19:40.752663Z","iopub.execute_input":"2026-03-04T09:19:40.753297Z","iopub.status.idle":"2026-03-04T09:20:19.209577Z","shell.execute_reply.started":"2026-03-04T09:19:40.753257Z","shell.execute_reply":"2026-03-04T09:20:19.208914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.210738Z","iopub.execute_input":"2026-03-04T09:20:19.211185Z","iopub.status.idle":"2026-03-04T09:20:19.215751Z","shell.execute_reply.started":"2026-03-04T09:20:19.211160Z","shell.execute_reply":"2026-03-04T09:20:19.215197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set to True because the preprocessing logic was changed (added motion features) \nPREPROCESS_DATA = True \nTRAIN_MODEL = True\nUSE_VAL = True\n\n# Mixed Precision on GPU P100 for faster training, 50% less VRAM usage\nfrom tensorflow.keras import mixed_precision\npolicy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\n\n# Data Dimensions\nN_ROWS = 543\nN_DIMS = 3\nDIM_NAMES = ['x', 'y', 'z']\nSEED = 42\nNUM_CLASSES = 250\nIS_INTERACTIVE = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '') == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\n# INPUT CONFIGURATION\n# Increased to 128 to capture fine temporal details (fingerspelling)\nINPUT_SIZE = 128 \n\n# Training Hyperparameters\nBATCH_ALL_SIGNS_N = 4\nBATCH_SIZE = 64\nN_EPOCHS = 200 # 100\nLR_MAX = 4e-4 #1e-3\nN_WARMUP_EPOCHS = 10 # Slight warmup helps Transformer stability\nWD_RATIO = 0.05\nMASK_VAL = 0.0 \n\n# Data type config\nDATA_DTYPE = np.float16\n\nprint(f'Compute Policy: {policy.compute_dtype}')\nprint(f'Variable Policy: {policy.variable_dtype}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.216723Z","iopub.execute_input":"2026-03-04T09:20:19.217007Z","iopub.status.idle":"2026-03-04T09:20:19.234888Z","shell.execute_reply.started":"2026-03-04T09:20:19.216978Z","shell.execute_reply":"2026-03-04T09:20:19.234208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.236455Z","iopub.execute_input":"2026-03-04T09:20:19.236747Z","iopub.status.idle":"2026-03-04T09:20:19.247935Z","shell.execute_reply.started":"2026-03-04T09:20:19.236725Z","shell.execute_reply":"2026-03-04T09:20:19.247284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read Training Data\n#if IS_INTERACTIVE or not PREPROCESS_DATA:\n#train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\n#else:\ntrain = pd.read_csv('/kaggle/input/competitions/asl-signs/train.csv')\n\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.248809Z","iopub.execute_input":"2026-03-04T09:20:19.249207Z","iopub.status.idle":"2026-03-04T09:20:19.471548Z","shell.execute_reply.started":"2026-03-04T09:20:19.249178Z","shell.execute_reply":"2026-03-04T09:20:19.470772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/competitions/asl-signs/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.472538Z","iopub.execute_input":"2026-03-04T09:20:19.472836Z","iopub.status.idle":"2026-03-04T09:20:19.516854Z","shell.execute_reply.started":"2026-03-04T09:20:19.472803Z","shell.execute_reply":"2026-03-04T09:20:19.516100Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encoding Labels","metadata":{}},{"cell_type":"code","source":"MAP_PATH = '/kaggle/input/competitions/asl-signs/sign_to_prediction_index_map.json'\n\nwith open(MAP_PATH, 'r') as f:\n    SIGN2ORD = json.load(f)\n\nORD2SIGN = {v: k for k, v in SIGN2ORD.items()}\n\ntrain['sign_ord'] = train['sign'].map(SIGN2ORD)\n\nassert train['sign_ord'].isna().sum() == 0, \"Mapping failed: Some signs in train are missing from the JSON map.\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.517944Z","iopub.execute_input":"2026-03-04T09:20:19.518272Z","iopub.status.idle":"2026-03-04T09:20:19.543262Z","shell.execute_reply.started":"2026-03-04T09:20:19.518243Z","shell.execute_reply":"2026-03-04T09:20:19.542543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train.sample(n=5)) \ndisplay(train.info()) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.544210Z","iopub.execute_input":"2026-03-04T09:20:19.544457Z","iopub.status.idle":"2026-03-04T09:20:19.594173Z","shell.execute_reply.started":"2026-03-04T09:20:19.544435Z","shell.execute_reply":"2026-03-04T09:20:19.593621Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"N = int(1e3) if (IS_INTERACTIVE or not PREPROCESS_DATA) else int(10e3)\nN_UNIQUE_FRAMES = np.zeros(N, dtype=np.uint16)\nN_MISSING_FRAMES = np.zeros(N, dtype=np.uint16)\nMAX_FRAME = np.zeros(N, dtype=np.uint16)\n\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999]\n\nfor idx, file_path in enumerate(tqdm(train['file_path'].sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    N_UNIQUE_FRAMES[idx] = df['frame'].nunique()\n    N_MISSING_FRAMES[idx] = (df['frame'].max() - df['frame'].min()) - df['frame'].nunique() + 1\n    MAX_FRAME[idx] = df['frame'].max() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:19.595069Z","iopub.execute_input":"2026-03-04T09:20:19.595342Z","iopub.status.idle":"2026-03-04T09:20:49.955356Z","shell.execute_reply.started":"2026-03-04T09:20:19.595319Z","shell.execute_reply":"2026-03-04T09:20:49.954539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_UNIQUE_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+25, 25))\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:49.957619Z","iopub.execute_input":"2026-03-04T09:20:49.957866Z","iopub.status.idle":"2026-03-04T09:20:50.363583Z","shell.execute_reply.started":"2026-03-04T09:20:49.957844Z","shell.execute_reply":"2026-03-04T09:20:50.362803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Number of missing frames, consecutive frames with missing intermediate frame, i.e. 1,2,4,5 but 3 is missing\ndisplay(pd.Series(N_MISSING_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_MISSING_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Missing Frames', size=24)\npd.Series(N_MISSING_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:50.364391Z","iopub.execute_input":"2026-03-04T09:20:50.364650Z","iopub.status.idle":"2026-03-04T09:20:50.607581Z","shell.execute_reply.started":"2026-03-04T09:20:50.364621Z","shell.execute_reply":"2026-03-04T09:20:50.606973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid() \nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:50.608347Z","iopub.execute_input":"2026-03-04T09:20:50.608604Z","iopub.status.idle":"2026-03-04T09:20:50.849833Z","shell.execute_reply.started":"2026-03-04T09:20:50.608571Z","shell.execute_reply":"2026-03-04T09:20:50.849298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid() \nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:50.850562Z","iopub.execute_input":"2026-03-04T09:20:50.850817Z","iopub.status.idle":"2026-03-04T09:20:51.305580Z","shell.execute_reply.started":"2026-03-04T09:20:50.850789Z","shell.execute_reply":"2026-03-04T09:20:51.304941Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Landmark Indices for Lips, Hands, and Pose","metadata":{}},{"cell_type":"code","source":"USE_TYPES = ['left_hand', 'pose', 'right_hand']\nSTART_IDX = 468\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n# Landmark indices in original data\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n# Landmark indices in processed data\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\nprint(f'# HAND_IDXS: {len(HAND_IDXS)}, N_COLS: {N_COLS}') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:51.306372Z","iopub.execute_input":"2026-03-04T09:20:51.306592Z","iopub.status.idle":"2026-03-04T09:20:51.315892Z","shell.execute_reply.started":"2026-03-04T09:20:51.306559Z","shell.execute_reply":"2026-03-04T09:20:51.315270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:51.316614Z","iopub.execute_input":"2026-03-04T09:20:51.316857Z","iopub.status.idle":"2026-03-04T09:20:51.333827Z","shell.execute_reply.started":"2026-03-04T09:20:51.316831Z","shell.execute_reply":"2026-03-04T09:20:51.333162Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Process data Tensorflow","metadata":{}},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:51.334562Z","iopub.execute_input":"2026-03-04T09:20:51.334843Z","iopub.status.idle":"2026-03-04T09:20:51.347885Z","shell.execute_reply.started":"2026-03-04T09:20:51.334813Z","shell.execute_reply":"2026-03-04T09:20:51.347202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__() \n        self.lips_idxs = LIPS_IDXS\n        self.left_hand_idxs = LEFT_HAND_IDXS\n        self.pose_idxs = POSE_IDXS\n        self.landmark_idxs_left = LANDMARK_IDXS_LEFT_DOMINANT0\n        self.landmark_idxs_right = LANDMARK_IDXS_RIGHT_DOMINANT0\n\n    @tf.function\n    def get_hand_distances(self, data):\n        # Extract the hand landmarks (Indices 40 to 60 in our gathered vector)\n        # 0: Wrist, 4: ThumbTip, 8: IndexTip, 12: MiddleTip, 16: RingTip, 20: PinkyTip\n        hand = data[:, 40:61, :] \n        wrist = hand[:, 0:1, :]\n        fingertips = tf.gather(hand, [4, 8, 12, 16, 20], axis=1)\n        \n        # Calculate Euclidean distance from wrist to each fingertip\n        # This explicitly encodes \"Hand Open\" vs \"Hand Closed\"\n        dists = tf.norm(fingertips - wrist, axis=-1) \n        return dists # Shape: (Frames, 5)\n\n    @tf.function\n    def call(self, data0):\n        # Cast and Initial Filter\n        data0 = tf.cast(data0, tf.float32)\n\n        # Dominant Hand Logic (Mirroring)\n        # We determine which hand has more valid data and standardize to \"Left Dominant\"\n        l_hand_nans = tf.reduce_sum(tf.cast(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), tf.int32))\n        r_hand_nans = tf.reduce_sum(tf.cast(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), tf.int32))\n        \n        left_dominant = l_hand_nans <= r_hand_nans\n        \n        if left_dominant:\n            data = tf.gather(data0, self.landmark_idxs_left, axis=1)\n        else:\n            data = tf.gather(data0, self.landmark_idxs_right, axis=1)\n            # Mirror X coordinate to make right-handed signers look left-handed\n            data = tf.concat([-1.0 * data[:, :, 0:1], data[:, :, 1:]], axis=-1)\n\n        # Filter empty frames based on dominant hand presence\n        hand_presence = tf.reduce_sum(tf.cast(tf.math.not_equal(tf.gather(data, range(40,61), axis=1), 0.0), tf.float32), axis=[1,2])\n        non_empty_idxs = tf.where(hand_presence > 0)\n        non_empty_idxs = tf.squeeze(non_empty_idxs, axis=1)\n        data = tf.gather(data, non_empty_idxs, axis=0)\n\n        # Normalization (Centering and Scaling)\n        # Center on Lips mean, scale by global standard deviation\n        lips = data[:, :40, :]\n        lips_mean = tf.math.reduce_mean(tf.where(tf.math.is_nan(lips), 0.0, lips), axis=1, keepdims=True)\n        data_std = tf.math.reduce_std(tf.where(tf.math.is_nan(data), 0.0, data), axis=[1,2], keepdims=True) + 1e-6\n        data = (data - lips_mean) / data_std\n        data = tf.where(tf.math.is_nan(data), 0.0, data)\n\n        # Temporal Interpolation to INPUT_SIZE (128)\n        n_frames = tf.shape(data)[0]\n        if n_frames < INPUT_SIZE:\n            # Padding\n            data = tf.pad(data, [[0, INPUT_SIZE - n_frames], [0,0], [0,0]], constant_values=0.0)\n            non_empty_frame_idxs = tf.pad(tf.cast(non_empty_idxs, tf.float32), [[0, INPUT_SIZE - n_frames]], constant_values=-1.0)\n        else:\n            # Bilinear Resizing\n            data = tf.reshape(data, [1, n_frames, -1, 1])\n            data = tf.image.resize(data, [INPUT_SIZE, tf.shape(data)[2]], method='bilinear')\n            data = tf.reshape(data, [INPUT_SIZE, -1, 3])\n            non_empty_frame_idxs = tf.linspace(0.0, tf.cast(n_frames, tf.float32), INPUT_SIZE)\n\n        # Feature Extraction: Hand Distances\n        # We calculate distances after normalization so they are scale-invariant\n        hand_dists = self.get_hand_distances(data) # (128, 5)\n\n        # Coordinate Flattening + Distances\n        # Concatenate XYZ (66*3=198) with Distances (5) = 203 features\n        data_flat = tf.reshape(data, (INPUT_SIZE, -1))\n        features = tf.concat([data_flat, hand_dists], axis=-1)\n\n        # Motion Features (dx and ddx)\n        # Captures velocity and acceleration of all features\n        dx = features[1:] - features[:-1]\n        dx = tf.concat([tf.zeros_like(features[:1]), dx], axis=0)\n        \n        ddx = dx[1:] - dx[:-1]\n        ddx = tf.concat([tf.zeros_like(dx[:1]), ddx], axis=0)\n\n        # Final Feature Stack: [X, dX, ddX]\n        # Total Features: 203 * 3 = 609 per frame\n        final_data = tf.concat([features, dx, ddx], axis=-1)\n        \n        return final_data, non_empty_frame_idxs\n\npreprocess_layer = PreprocessLayer()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:51.348808Z","iopub.execute_input":"2026-03-04T09:20:51.349030Z","iopub.status.idle":"2026-03-04T09:20:51.371370Z","shell.execute_reply.started":"2026-03-04T09:20:51.349010Z","shell.execute_reply":"2026-03-04T09:20:51.370764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n        \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:51.372358Z","iopub.execute_input":"2026-03-04T09:20:51.373060Z","iopub.status.idle":"2026-03-04T09:20:51.387727Z","shell.execute_reply.started":"2026-03-04T09:20:51.373036Z","shell.execute_reply":"2026-03-04T09:20:51.387025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Memory Optimized: Split Metadata FIRST to avoid \"Copy Spike\" OOM\ndef preprocess_data():\n    # Infer Feature Shape from dummy data\n    print(\"Determining feature shape...\")\n    dummy_path = train['file_path'].values[0]\n    dummy_data, _ = get_data(dummy_path)\n    N_COLS_FINAL = dummy_data.shape[1]\n    print(f\"Features per frame: {N_COLS_FINAL}\")\n\n    # Synchronize labels with the official JSON mapping\n    train['sign_ord'] = train['sign'].map(SIGN2ORD)\n\n    # Split metadata by participant_id to prevent data leakage between sets\n    print(\"Splitting metadata by participant...\")\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=1, random_state=SEED)\n    train_idxs, val_idxs = next(splitter.split(train, groups=train['participant_id']))\n    \n    df_train = train.iloc[train_idxs]\n    df_val = train.iloc[val_idxs]\n    \n    print(f\"Train Samples: {len(df_train)} | Val Samples: {len(df_val)}\")\n\n    # Declare global variables for training access\n    global X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN\n    global X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL, validation_data\n\n    # Allocate X_train as float16 immediately to save 50% VRAM on P100\n    X_train = np.zeros([len(df_train), INPUT_SIZE, N_COLS_FINAL], dtype=np.float16)\n    y_train = np.zeros([len(df_train)], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.full([len(df_train), INPUT_SIZE], -1, dtype=np.float16)\n    \n    print(\"Filling X_train...\")\n    for i, (file_path, sign_ord) in enumerate(tqdm(df_train[['file_path', 'sign_ord']].values)):\n        # Explicit garbage collection to maintain memory headroom\n        if i % 5000 == 0: gc.collect()\n        \n        data, non_empty_frame_idxs = get_data(file_path)\n        \n        # np.nan_to_num is mandatory; LSTMs are highly unstable with NaN inputs\n        X_train[i] = np.nan_to_num(data.numpy().astype(np.float16))\n        y_train[i] = sign_ord\n        NON_EMPTY_FRAME_IDXS_TRAIN[i] = non_empty_frame_idxs.numpy().astype(np.float16)\n\n    if USE_VAL:\n        print(\"Processing Validation Data...\")\n        X_val = np.zeros([len(df_val), INPUT_SIZE, N_COLS_FINAL], dtype=np.float16)\n        y_val = np.zeros([len(df_val)], dtype=np.int32)\n        NON_EMPTY_FRAME_IDXS_VAL = np.full([len(df_val), INPUT_SIZE], -1, dtype=np.float16)\n        \n        for i, (file_path, sign_ord) in enumerate(tqdm(df_val[['file_path', 'sign_ord']].values)):\n            data, non_empty_frame_idxs = get_data(file_path)\n            \n            X_val[i] = np.nan_to_num(data.numpy().astype(np.float16))\n            y_val[i] = sign_ord\n            NON_EMPTY_FRAME_IDXS_VAL[i] = non_empty_frame_idxs.numpy().astype(np.float16)\n\n        # Validation data setup for model.fit\n        y_val_oh = tf.one_hot(y_val, NUM_CLASSES)\n        validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val_oh)\n    else:\n        validation_data = None\n\n    return N_COLS_FINAL\n\n# Execute Preprocessing\nif PREPROCESS_DATA:\n    N_COLS_FINAL = preprocess_data()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:20:51.388733Z","iopub.execute_input":"2026-03-04T09:20:51.388942Z","iopub.status.idle":"2026-03-04T10:01:42.135006Z","shell.execute_reply.started":"2026-03-04T09:20:51.388922Z","shell.execute_reply":"2026-03-04T10:01:42.134266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data Verification: Checks the in-memory arrays\n# Verify N_COLS_FINAL is set\nif 'N_COLS_FINAL' not in locals():\n    # Fallback inference\n    N_COLS_FINAL = X_train.shape[2]\n\nprint(f\"Features per frame (N_COLS_FINAL): {N_COLS_FINAL}\")\nprint(f\"X_train shape: {X_train.shape}  | dtype: {X_train.dtype}\")\nprint(f\"y_train shape: {y_train.shape}  | dtype: {y_train.dtype}\")\nprint(\"-\" * 30)\n\nif USE_VAL:\n    print(f\"X_val shape:   {X_val.shape}    | dtype: {X_val.dtype}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:42.136202Z","iopub.execute_input":"2026-03-04T10:01:42.136483Z","iopub.status.idle":"2026-03-04T10:01:42.141493Z","shell.execute_reply.started":"2026-03-04T10:01:42.136456Z","shell.execute_reply":"2026-03-04T10:01:42.140880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class Count\ndisplay(pd.Series(y_train).value_counts().to_frame('Class Count').iloc[[0,1,2,3,4, -5,-4,-3,-2,-1]]) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:42.142410Z","iopub.execute_input":"2026-03-04T10:01:42.142764Z","iopub.status.idle":"2026-03-04T10:01:42.184082Z","shell.execute_reply.started":"2026-03-04T10:01:42.142737Z","shell.execute_reply":"2026-03-04T10:01:42.183449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Vast majority of samples fits has less than 32 non empty frames\nN_EMPTY_FRAMES = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum(axis=1) \nN_EMPTY_FRAMES_WATERFALL = []\nfor n in tqdm(range(1,INPUT_SIZE+1)):\n    N_EMPTY_FRAMES_WATERFALL.append(sum(N_EMPTY_FRAMES >= n) / len(NON_EMPTY_FRAME_IDXS_TRAIN) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Non Empty Frames')\npd.Series(N_EMPTY_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks(np.arange(INPUT_SIZE), np.arange(1, INPUT_SIZE+1))\nplt.xlabel('Number of Non Empty Frames', size=16)\nplt.yticks(np.arange(0, 100+10, 10))\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Non Empty Frames', size=16)\nplt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:42.185116Z","iopub.execute_input":"2026-03-04T10:01:42.185427Z","iopub.status.idle":"2026-03-04T10:01:44.043053Z","shell.execute_reply.started":"2026-03-04T10:01:42.185399Z","shell.execute_reply":"2026-03-04T10:01:44.042403Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Samples","metadata":{}},{"cell_type":"code","source":"# Features: Standard Batching + Spatial Augs + Light Masking + Low FPS + Noise\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, batch_size=32, mixup_alpha=0.0):\n    \"\"\"\n    Generator for training batches. MixUp is disabled (0.0) and masking is reduced \n    to prevent starving the pure Transformer of meaningful sequence patterns.\n    \"\"\"\n    n_samples = len(X)\n    indices = np.arange(n_samples)\n    \n    while True:\n        np.random.shuffle(indices)\n        \n        for start_idx in range(0, n_samples, batch_size):\n            end_idx = min(start_idx + batch_size, n_samples)\n            if end_idx - start_idx < 4: \n                continue\n                \n            batch_idxs = indices[start_idx:end_idx]\n            current_batch_size = len(batch_idxs)\n            \n            X_batch = X[batch_idxs].astype(np.float32) \n            y_batch = tf.one_hot(y[batch_idxs], NUM_CLASSES).numpy()\n            non_empty_frame_idxs_batch = NON_EMPTY_FRAME_IDXS[batch_idxs].copy()\n\n            # Spatial Augmentations (Vectorized)\n            if np.random.rand() < 0.5:\n                scale = np.random.uniform(0.8, 1.2, size=(current_batch_size, 1, 1)).astype(np.float32)\n                X_batch = X_batch * scale\n                \n            if np.random.rand() < 0.5:\n                shift = np.random.uniform(-0.05, 0.05, size=(current_batch_size, 1, 1)).astype(np.float32)\n                X_batch = X_batch + shift\n\n            # Gaussian Noise\n            if np.random.rand() < 0.4:\n                X_batch += np.random.normal(0, 0.012, X_batch.shape)\n\n            # Temporal & Feature Masking (Reduced Probabilities)\n            for b in range(current_batch_size):\n                \n                # Feature Masking (Reduced to 10%)\n                if np.random.rand() < 0.1:\n                    mask_indices = np.random.choice(X.shape[2], int(X.shape[2]*0.1), replace=False)\n                    X_batch[b, :, mask_indices] = 0.0\n\n                # Time Masking (Reduced to 20% chance)\n                if np.random.rand() < 0.2:\n                    len_mask = np.random.randint(5, 15)\n                    start_mask = np.random.randint(0, INPUT_SIZE - len_mask)\n                    X_batch[b, start_mask:start_mask+len_mask, :] = 0.0\n                    non_empty_frame_idxs_batch[b, start_mask:start_mask+len_mask] = -1\n                \n                # Low FPS Simulation (Reduced to 20% chance)\n                if np.random.rand() < 0.2:\n                    step = np.random.randint(2, 4) \n                    frame_indices = np.arange(0, INPUT_SIZE, step)\n                    \n                    kept_frames = X_batch[b, frame_indices, :]\n                    kept_masks = non_empty_frame_idxs_batch[b, frame_indices]\n                    \n                    X_batch[b] = 0.0\n                    non_empty_frame_idxs_batch[b] = -1\n                    \n                    valid_len = len(kept_frames)\n                    X_batch[b, :valid_len, :] = kept_frames\n                    non_empty_frame_idxs_batch[b, :valid_len] = kept_masks\n\n            # MixUp (Currently bypassed since mixup_alpha=0.0)\n            if mixup_alpha > 0 and np.random.rand() < 0.5:\n                lam = np.random.beta(mixup_alpha, mixup_alpha)\n                perm_indices = np.random.permutation(current_batch_size)\n                \n                X_batch = lam * X_batch + (1 - lam) * X_batch[perm_indices]\n                y_batch = lam * y_batch + (1 - lam) * y_batch[perm_indices]\n                \n                if lam < 0.5:\n                    non_empty_frame_idxs_batch = non_empty_frame_idxs_batch[perm_indices]\n            \n            yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:44.044012Z","iopub.execute_input":"2026-03-04T10:01:44.044299Z","iopub.status.idle":"2026-03-04T10:01:44.055972Z","shell.execute_reply.started":"2026-03-04T10:01:44.044276Z","shell.execute_reply":"2026-03-04T10:01:44.055475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN)\nX_batch, y_batch = next(dummy_dataset)\n\nfor k, v in X_batch.items():\n    print(f'{k} shape: {v.shape}, dtype: {v.dtype}')\n\n# Batch shape/dtype\nprint(f'y_batch shape: {y_batch.shape}, dtype: {y_batch.dtype}')\n# Verify each batch contains each sign exactly N times\ndisplay(pd.Series(np.argmax(y_batch, axis=1)).value_counts().to_frame('Counts')) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:44.056741Z","iopub.execute_input":"2026-03-04T10:01:44.057000Z","iopub.status.idle":"2026-03-04T10:01:44.093875Z","shell.execute_reply.started":"2026-03-04T10:01:44.056972Z","shell.execute_reply":"2026-03-04T10:01:44.093389Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Config","metadata":{}},{"cell_type":"code","source":"embed_dim = 256\nnum_heads = 4\nff_dim = embed_dim * 2\n\nclass LearnablePositionalEmbedding(tf.keras.layers.Layer):\n    def __init__(self, max_len, embed_dim, **kwargs):\n        super().__init__(**kwargs)\n        self.pos_embedding = tf.keras.layers.Embedding(input_dim=max_len, output_dim=embed_dim)\n\n    def call(self, x):\n        max_len = tf.shape(x)[1]\n        positions = tf.range(start=0, limit=max_len, delta=1)\n        return x + self.pos_embedding(positions)\n\nclass StochasticDepth(tf.keras.layers.Layer):\n    def __init__(self, drop_prob=0.2, **kwargs):\n        super().__init__(**kwargs)\n        self.drop_prob = drop_prob\n\n    def call(self, x, training=None):\n        if not training or self.drop_prob == 0.:\n            return x\n        \n        keep_prob = 1.0 - self.drop_prob\n        # Graph-safe shape evaluation\n        shape = (tf.shape(x)[0],) + (1,) * (len(x.shape) - 1)\n        random_tensor = keep_prob + tf.random.uniform(shape, dtype=x.dtype)\n        binary_tensor = tf.floor(random_tensor)\n        \n        return (x / keep_prob) * binary_tensor\n\nclass TransformerBlock(tf.keras.layers.Layer):\n    def __init__(self, embed_dim, num_heads, ff_dim, drop_rate=0.1, drop_path_rate=0.1, **kwargs):\n        super().__init__(**kwargs)\n        \n        # Proper dimension splitting per head\n        head_dim = embed_dim // num_heads \n        self.att = tf.keras.layers.MultiHeadAttention(num_heads=num_heads, key_dim=head_dim)\n        \n        # Conformer Trick: Local temporal awareness\n        self.local_conv = tf.keras.layers.DepthwiseConv1D(kernel_size=5, padding='same', use_bias=False)\n        \n        self.ffn = tf.keras.Sequential([\n            tf.keras.layers.Dense(ff_dim, activation=\"swish\"), \n            tf.keras.layers.Dense(embed_dim),\n        ])\n        \n        self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        \n        self.dropout1 = tf.keras.layers.Dropout(drop_rate)\n        self.dropout2 = tf.keras.layers.Dropout(drop_rate)\n        self.drop_path = StochasticDepth(drop_path_rate)\n\n    def call(self, inputs, training=None):\n        # Global Attention Path\n        norm_inputs = self.layernorm1(inputs)\n        attn_output = self.att(norm_inputs, norm_inputs, training=training)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = inputs + self.drop_path(attn_output, training=training)\n        \n        # Local Convolution + FFN Path\n        norm_out1 = self.layernorm2(out1)\n        local_features = self.local_conv(norm_out1) \n        \n        ffn_output = self.ffn(local_features, training=training)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        \n        return out1 + self.drop_path(ffn_output, training=training)\n\nclass MaskingLayer(tf.keras.layers.Layer):\n    def __init__(self, **kwargs):\n        super(MaskingLayer, self).__init__(**kwargs)\n    \n    def call(self, inputs):\n        frames, non_empty_frame_idxs = inputs\n        mask = tf.math.not_equal(non_empty_frame_idxs, -1)\n        mask = tf.cast(mask, dtype=frames.dtype)\n        return frames * tf.expand_dims(mask, -1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:44.094668Z","iopub.execute_input":"2026-03-04T10:01:44.094869Z","iopub.status.idle":"2026-03-04T10:01:44.106594Z","shell.execute_reply.started":"2026-03-04T10:01:44.094839Z","shell.execute_reply":"2026-03-04T10:01:44.105997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model():\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS_FINAL], dtype=tf.float16, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float16, name='non_empty_frame_idxs')\n    \n    x = MaskingLayer(name='input_masking')([frames, non_empty_frame_idxs])\n    \n    x = tf.keras.layers.Dense(embed_dim, use_bias=False, name='stem_projection')(x)\n    x = tf.keras.layers.LayerNormalization(epsilon=1e-6, name='stem_ln')(x)\n    \n    x = LearnablePositionalEmbedding(INPUT_SIZE, embed_dim)(x)\n    \n    # Relaxed Stochastic Depth\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, drop_rate=0.1, drop_path_rate=0.00, name='transformer_block_1')(x)\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, drop_rate=0.1, drop_path_rate=0.05, name='transformer_block_2')(x)\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, drop_rate=0.1, drop_path_rate=0.05, name='transformer_block_3')(x)\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, drop_rate=0.1, drop_path_rate=0.10, name='transformer_block_4')(x)\n    \n    x = tf.keras.layers.GlobalAveragePooling1D(name='gap')(x)\n    \n    # Reduced Late Dropout\n    x = tf.keras.layers.Dropout(0.4, name='late_dropout')(x) \n    \n    outputs = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax', dtype='float32', name='classifier')(x)\n    \n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    optimizer = tf.keras.optimizers.AdamW(learning_rate=LR_MAX, weight_decay=WD_RATIO, clipnorm=1.0)\n    loss = tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1)\n    \n    metrics = [\n        tf.keras.metrics.CategoricalAccuracy(name='acc'),\n        tf.keras.metrics.TopKCategoricalAccuracy(k=5, name='top_5_acc'),\n    ]\n    \n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model\n\ntf.keras.backend.clear_session()\nmodel = get_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:44.107366Z","iopub.execute_input":"2026-03-04T10:01:44.107549Z","iopub.status.idle":"2026-03-04T10:01:47.493918Z","shell.execute_reply.started":"2026-03-04T10:01:44.107533Z","shell.execute_reply":"2026-03-04T10:01:47.493404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:47.494747Z","iopub.execute_input":"2026-03-04T10:01:47.494959Z","iopub.status.idle":"2026-03-04T10:01:48.766404Z","shell.execute_reply.started":"2026-03-04T10:01:47.494939Z","shell.execute_reply":"2026-03-04T10:01:48.765603Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# No NaN Predictions","metadata":{}},{"cell_type":"code","source":"if TRAIN_MODEL:\n    y_pred = model.predict_on_batch(X_batch)\n    nan_count = np.isnan(y_pred).sum()\n    print(f'# NaN Values In Prediction: {nan_count}')\n    if nan_count > 0:\n        raise ValueError(\"Stability Error: NaNs detected in initial forward pass. Check normalization.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:48.769821Z","iopub.execute_input":"2026-03-04T10:01:48.770053Z","iopub.status.idle":"2026-03-04T10:01:54.188486Z","shell.execute_reply.started":"2026-03-04T10:01:48.770032Z","shell.execute_reply":"2026-03-04T10:01:54.187760Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Weight Initialization","metadata":{}},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plt.figure(figsize=(12,5))\n    plt.title(f'Softmax Output Distribution | µ={y_pred.mean():.3f}, σ={y_pred.std():.3f}', pad=25)\n    pd.Series(y_pred.flatten()).plot(kind='hist', bins=128, label='Class Probability')\n    plt.vlines([1 / NUM_CLASSES], 0, plt.ylim()[1], color='red', label='Random Guess Baseline')\n    plt.grid()\n    plt.legend()\n    plt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:54.189362Z","iopub.execute_input":"2026-03-04T10:01:54.189568Z","iopub.status.idle":"2026-03-04T10:01:54.478267Z","shell.execute_reply.started":"2026-03-04T10:01:54.189548Z","shell.execute_reply":"2026-03-04T10:01:54.477653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if USE_VAL:\n    # Verify validation distribution covers the full 250-sign vocabulary\n    unique_val_signs = pd.Series(y_val).nunique()\n    print(f'# Unique Signs in Validation Set: {unique_val_signs}')\n    if unique_val_signs < NUM_CLASSES:\n        print(\"Warning: Validation set is missing signs\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:54.479187Z","iopub.execute_input":"2026-03-04T10:01:54.479493Z","iopub.status.idle":"2026-03-04T10:01:54.486390Z","shell.execute_reply.started":"2026-03-04T10:01:54.479463Z","shell.execute_reply":"2026-03-04T10:01:54.485713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Memory Optimization: Generator to stream validation data without a \"Copy Spike\" OOM.\ndef get_val_batch(X, y, NON_EMPTY_FRAME_IDXS, batch_size=32):\n    n_samples = len(X)\n    indices = np.arange(n_samples)\n    while True:\n        for start_idx in range(0, n_samples, batch_size):\n            end_idx = min(start_idx + batch_size, n_samples)\n            batch_idxs = indices[start_idx:end_idx]\n            X_batch = X[batch_idxs].astype(np.float32)\n            y_batch = tf.one_hot(y[batch_idxs], NUM_CLASSES).numpy()\n            non_empty_frame_idxs_batch = NON_EMPTY_FRAME_IDXS[batch_idxs]\n            yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:54.487272Z","iopub.execute_input":"2026-03-04T10:01:54.487500Z","iopub.status.idle":"2026-03-04T10:01:54.500776Z","shell.execute_reply.started":"2026-03-04T10:01:54.487475Z","shell.execute_reply":"2026-03-04T10:01:54.500277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Synchronized Training Engine\n# Optimized for pure Transformers: Requires Linear Warmup + Cosine Decay\n\nSTEPS_PER_EPOCH = len(X_train) // BATCH_SIZE\nVAL_STEPS = len(X_val) // BATCH_SIZE if USE_VAL else None\nTOTAL_STEPS = N_EPOCHS * STEPS_PER_EPOCH\nWARMUP_STEPS = N_WARMUP_EPOCHS * STEPS_PER_EPOCH\n\nclass WarmUpCosineDecay(tf.keras.optimizers.schedules.LearningRateSchedule):\n    \"\"\"\n    Linearly increases learning rate for WARMUP_STEPS, \n    then applies cosine decay to a minimum learning rate.\n    \"\"\"\n    def __init__(self, init_lr, max_lr, warmup_steps, total_steps):\n        super().__init__()\n        self.init_lr = tf.cast(init_lr, tf.float32)\n        self.max_lr = tf.cast(max_lr, tf.float32)\n        self.warmup_steps = tf.cast(warmup_steps, tf.float32)\n        self.total_steps = tf.cast(total_steps, tf.float32)\n\n    def __call__(self, step):\n        step = tf.cast(step, tf.float32)\n        \n        # Linear Warmup Phase\n        warmup_lr = self.init_lr + (self.max_lr - self.init_lr) * (step / self.warmup_steps)\n        \n        # Cosine Decay Phase\n        decay_steps = tf.maximum(self.total_steps - self.warmup_steps, 1.0)\n        step_after_warmup = tf.maximum(step - self.warmup_steps, 0.0)\n        cosine_decay = 0.5 * (1.0 + tf.math.cos(math.pi * step_after_warmup / decay_steps))\n        decayed_lr = cosine_decay * self.max_lr\n        \n        return tf.where(step < self.warmup_steps, warmup_lr, decayed_lr)\n\n# Initialize custom learning rate schedule\nlr_schedule = WarmUpCosineDecay(\n    init_lr=1e-6, \n    max_lr=LR_MAX, \n    warmup_steps=WARMUP_STEPS, \n    total_steps=TOTAL_STEPS\n)\n\n# AdamW Optimizer with Weight Decay\noptimizer = tf.keras.optimizers.AdamW(\n    learning_rate=lr_schedule, \n    weight_decay=WD_RATIO, \n    clipnorm=1.0 # Protects against gradient spikes in mixed_float16\n)\n\nmodel.compile(\n    loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.1),\n    optimizer=optimizer,\n    metrics=['acc', tf.keras.metrics.TopKCategoricalAccuracy(k=5, name='top_5_acc')]\n)\n\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss', \n        patience=25, # Transformers often need longer patience to settle\n        restore_best_weights=True,\n        verbose=1\n    ),\n    tf.keras.callbacks.ModelCheckpoint(\n        'transformer_model.weights.h5', \n        save_best_only=True, \n        save_weights_only=True\n    )\n]\n\nif TRAIN_MODEL:\n    history = model.fit(\n        # Training Generator (Applies spatial/temporal augs on CPU)\n        x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN, batch_size=BATCH_SIZE),\n        steps_per_epoch=STEPS_PER_EPOCH,\n        \n        # Validation Generator (Streamed to avoid OOM on P100)\n        validation_data=get_val_batch(X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL, batch_size=BATCH_SIZE) if USE_VAL else None,\n        validation_steps=VAL_STEPS if USE_VAL else None,\n        \n        epochs=N_EPOCHS,\n        callbacks=callbacks,\n        verbose=VERBOSE\n    ) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T10:01:54.501677Z","iopub.execute_input":"2026-03-04T10:01:54.501898Z","iopub.status.idle":"2026-03-04T13:58:41.174511Z","shell.execute_reply.started":"2026-03-04T10:01:54.501879Z","shell.execute_reply":"2026-03-04T13:58:41.173750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom IPython.display import FileLink\n\n# history.history is a dictionary containing the metrics\nhistory_df = pd.DataFrame(history.history)\n\n# Name the index 'epoch' for clarity\nhistory_df.index.name = 'epoch'\n\n# Save to CSV\nfile_name = 'training_history_post_train.csv'\nhistory_df.to_csv(file_name)\n\nprint(f\"Training history saved to {file_name}\")\n\n# Create a download link\ndisplay(FileLink(file_name))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:58:41.175866Z","iopub.execute_input":"2026-03-04T13:58:41.176135Z","iopub.status.idle":"2026-03-04T13:58:41.208449Z","shell.execute_reply.started":"2026-03-04T13:58:41.176111Z","shell.execute_reply":"2026-03-04T13:58:41.207866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Aggressive cleanup is necessary on P100 to prepare for the high-memory inference pass.\nif 'X_train' in globals():\n    del X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN\ngc.collect()\ntry:\n    import ctypes\n    ctypes.CDLL(\"libc.so.6\").malloc_trim(0)\n    print(\"System memory explicitly released.\")\nexcept:\n    pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:58:41.209261Z","iopub.execute_input":"2026-03-04T13:58:41.209578Z","iopub.status.idle":"2026-03-04T13:58:41.918290Z","shell.execute_reply.started":"2026-03-04T13:58:41.209556Z","shell.execute_reply":"2026-03-04T13:58:41.917665Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation - Comprehensive Report for 250 Classes","metadata":{}},{"cell_type":"code","source":"# Streamed Inference: Decouples prediction from RAM limits.\nif USE_VAL:\n    print(\"Running inference on validation set...\")\n    \n    y_val_pred = np.zeros(len(X_val), dtype=np.int16)\n    y_val_probs = np.zeros((len(X_val), 250), dtype=np.float32)  \n    chunk_size = 1000 \n    \n    for i in tqdm(range(int(np.ceil(len(X_val) / chunk_size))), desc=\"Inference\"):\n        start, end = i * chunk_size, min((i + 1) * chunk_size, len(X_val))\n        probs = model.predict(\n            {'frames': X_val[start:end], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL[start:end]}, \n            batch_size=BATCH_SIZE, verbose=0\n        )\n        y_val_probs[start:end] = probs  # STORE PROBABILITIES\n        y_val_pred[start:end] = probs.argmax(axis=1).astype(np.int16)\n        del probs\n        gc.collect() \n    \n    print(f\"Inference complete. Predictions shape: {y_val_pred.shape}\")\n    print(f\"Probabilities shape: {y_val_probs.shape}\")  # VERIFY","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:58:41.919124Z","iopub.execute_input":"2026-03-04T13:58:41.919524Z","iopub.status.idle":"2026-03-04T13:59:19.278924Z","shell.execute_reply.started":"2026-03-04T13:58:41.919499Z","shell.execute_reply":"2026-03-04T13:59:19.278171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Top-1 vs Top-5 Accuracy","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Top-1 Accuracy\n    top1_accuracy = sklearn.metrics.accuracy_score(y_val, y_val_pred)\n    \n    # Top-5 Accuracy\n    top5_preds = np.argsort(y_val_probs, axis=1)[:, -5:]  # Get indices of top 5 predictions\n    top5_correct = np.array([y_val[i] in top5_preds[i] for i in range(len(y_val))])\n    top5_accuracy = top5_correct.mean()\n    \n    print(\"=\"*60)\n    print(\"ACCURACY METRICS FOR 250-CLASS CLASSIFICATION\")\n    print(\"=\"*60)\n    print(f\"Top-1 Accuracy: {top1_accuracy:.4f} ({top1_accuracy*100:.2f}%)\")\n    print(f\"Top-5 Accuracy: {top5_accuracy:.4f} ({top5_accuracy*100:.2f}%)\")\n    print(f\"\\nInterpretation:\")\n    print(f\"  - The model predicts the exact correct sign {top1_accuracy*100:.2f}% of the time\")\n    print(f\"  - The correct sign appears in the model's top 5 guesses {top5_accuracy*100:.2f}% of the time\")\n    print(f\"  - Top-5 improvement: +{(top5_accuracy - top1_accuracy)*100:.2f} percentage points\")\n    print(\"=\"*60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:19.280207Z","iopub.execute_input":"2026-03-04T13:59:19.280470Z","iopub.status.idle":"2026-03-04T13:59:19.385978Z","shell.execute_reply.started":"2026-03-04T13:59:19.280447Z","shell.execute_reply":"2026-03-04T13:59:19.385443Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Confusion Matrix - Top 10 Most Confused Pairs","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Compute full confusion matrix\n    from sklearn.metrics import confusion_matrix\n    \n    cm = confusion_matrix(y_val, y_val_pred)\n    \n    # Find top 10 most confused pairs (excluding diagonal)\n    cm_no_diag = cm.copy()\n    np.fill_diagonal(cm_no_diag, 0)\n    \n    # Get indices of top confusions\n    top_confusions = []\n    for i in range(NUM_CLASSES):\n        for j in range(NUM_CLASSES):\n            if i != j and cm_no_diag[i, j] > 0:\n                top_confusions.append({\n                    'true_class': i,\n                    'pred_class': j,\n                    'count': cm_no_diag[i, j],\n                    'true_sign': ORD2SIGN[i],\n                    'pred_sign': ORD2SIGN[j]\n                })\n    \n    # Sort by count and get top 10\n    top_confusions = sorted(top_confusions, key=lambda x: x['count'], reverse=True)[:10]\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"TOP 10 MOST CONFUSED SIGN PAIRS\")\n    print(\"=\"*80)\n    \n    confusion_df = pd.DataFrame(top_confusions)\n    confusion_df['confusion_pair'] = confusion_df.apply(\n        lambda row: f\"{row['true_sign']} → {row['pred_sign']}\", axis=1\n    )\n    \n    display(confusion_df[['confusion_pair', 'count', 'true_sign', 'pred_sign']])\n    \n    # Plot confusion heatmap for top confused pairs\n    if len(top_confusions) > 0:\n        # Get unique classes involved in top confusions\n        confused_classes = set()\n        for conf in top_confusions:\n            confused_classes.add(conf['true_class'])\n            confused_classes.add(conf['pred_class'])\n        confused_classes = sorted(list(confused_classes))[:20]  # Limit to 20 for readability\n        \n        # Create sub-confusion matrix\n        cm_subset = cm[np.ix_(confused_classes, confused_classes)]\n        labels_subset = [ORD2SIGN[i] for i in confused_classes]\n        \n        plt.figure(figsize=(14, 12))\n        sns.heatmap(cm_subset, annot=True, fmt='d', cmap='YlOrRd', \n                    xticklabels=labels_subset, yticklabels=labels_subset,\n                    cbar_kws={'label': 'Number of Predictions'})\n        plt.title('Confusion Matrix: Most Frequently Confused Signs', fontsize=16, pad=20)\n        plt.xlabel('Predicted Sign', fontsize=14)\n        plt.ylabel('True Sign', fontsize=14)\n        plt.xticks(rotation=45, ha='right')\n        plt.yticks(rotation=0)\n        plt.tight_layout()\n        plt.show()\n        \n        print(\"\\nInsights:\")\n        print(\"  - Look for pairs with similar handshapes or movements\")\n        print(\"  - Symmetric confusions may indicate visual similarity\")\n        print(\"  - High confusion counts suggest challenging distinction tasks\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:19.389147Z","iopub.execute_input":"2026-03-04T13:59:19.389443Z","iopub.status.idle":"2026-03-04T13:59:20.204328Z","shell.execute_reply.started":"2026-03-04T13:59:19.389421Z","shell.execute_reply":"2026-03-04T13:59:20.203623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_classification_report():\n    report_labels = [ORD2SIGN[i] for i in range(NUM_CLASSES)]\n    report = sklearn.metrics.classification_report(\n        y_val, y_val_pred, target_names=report_labels, output_dict=True\n    )\n    df_report = pd.DataFrame(report).T.round(2)\n    display(df_report.sort_values('f1-score', ascending=False).head(20))\n    display(df_report.tail(3))\n\nif USE_VAL:\n    print_classification_report() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:20.205157Z","iopub.execute_input":"2026-03-04T13:59:20.205434Z","iopub.status.idle":"2026-03-04T13:59:20.246309Z","shell.execute_reply.started":"2026-03-04T13:59:20.205411Z","shell.execute_reply":"2026-03-04T13:59:20.245758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Per-Class F1 Score - Best and Worst Performing Signs","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    from sklearn.metrics import classification_report, f1_score\n    \n    # Get per-class F1 scores\n    report_labels = [ORD2SIGN[i] for i in range(NUM_CLASSES)]\n    report = classification_report(\n        y_val, y_val_pred, target_names=report_labels, output_dict=True, zero_division=0\n    )\n    \n    # Convert to DataFrame for easier manipulation\n    df_report = pd.DataFrame(report).T\n    \n    # Filter only class-level metrics (exclude macro/micro/weighted averages)\n    class_metrics = df_report[~df_report.index.isin(['accuracy', 'macro avg', 'micro avg', 'weighted avg'])].copy()\n    class_metrics = class_metrics.sort_values('f1-score', ascending=False)\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"TOP 5 BEST PERFORMING SIGNS (Highest F1 Score)\")\n    print(\"=\"*80)\n    best_5 = class_metrics.head(5)\n    best_5_display = best_5[['precision', 'recall', 'f1-score', 'support']].round(4)\n    display(best_5_display)\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"TOP 5 WORST PERFORMING SIGNS (Lowest F1 Score)\")\n    print(\"=\"*80)\n    worst_5 = class_metrics.tail(5)\n    worst_5_display = worst_5[['precision', 'recall', 'f1-score', 'support']].round(4)\n    display(worst_5_display)\n    \n    # Visualize F1 score distribution\n    plt.figure(figsize=(16, 6))\n    \n    # Histogram of F1 scores\n    plt.subplot(1, 2, 1)\n    plt.hist(class_metrics['f1-score'], bins=30, edgecolor='black', alpha=0.7)\n    plt.axvline(class_metrics['f1-score'].mean(), color='red', linestyle='--', \n                label=f'Mean F1: {class_metrics[\"f1-score\"].mean():.3f}')\n    plt.axvline(class_metrics['f1-score'].median(), color='green', linestyle='--', \n                label=f'Median F1: {class_metrics[\"f1-score\"].median():.3f}')\n    plt.xlabel('F1 Score', fontsize=12)\n    plt.ylabel('Number of Signs', fontsize=12)\n    plt.title('Distribution of Per-Class F1 Scores', fontsize=14)\n    plt.legend()\n    plt.grid(True, alpha=0.3)\n    \n    # Box plot of F1 scores\n    plt.subplot(1, 2, 2)\n    plt.boxplot(class_metrics['f1-score'], vert=True)\n    plt.ylabel('F1 Score', fontsize=12)\n    plt.title('F1 Score Distribution (Box Plot)', fontsize=14)\n    plt.grid(True, alpha=0.3)\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Summary statistics\n    print(\"\\n\" + \"=\"*80)\n    print(\"F1 SCORE STATISTICS ACROSS ALL 250 CLASSES\")\n    print(\"=\"*80)\n    print(f\"Mean F1 Score:   {class_metrics['f1-score'].mean():.4f}\")\n    print(f\"Median F1 Score: {class_metrics['f1-score'].median():.4f}\")\n    print(f\"Std Dev:         {class_metrics['f1-score'].std():.4f}\")\n    print(f\"Min F1 Score:    {class_metrics['f1-score'].min():.4f}\")\n    print(f\"Max F1 Score:    {class_metrics['f1-score'].max():.4f}\")\n    print(f\"\\nClasses with F1 = 0.0: {(class_metrics['f1-score'] == 0).sum()}\")\n    print(f\"Classes with F1 > 0.9: {(class_metrics['f1-score'] > 0.9).sum()}\")\n    print(f\"Classes with F1 > 0.8: {(class_metrics['f1-score'] > 0.8).sum()}\")\n    print(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:20.247068Z","iopub.execute_input":"2026-03-04T13:59:20.247368Z","iopub.status.idle":"2026-03-04T13:59:20.598530Z","shell.execute_reply.started":"2026-03-04T13:59:20.247344Z","shell.execute_reply":"2026-03-04T13:59:20.597867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Analysis: Why Does the Model Fail?","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Analyze characteristics of best and worst performing signs\n    print(\"\\n\" + \"=\"*80)\n    print(\"FAILURE ANALYSIS\")\n    print(\"=\"*80)\n    \n    # Get sample counts for worst performing signs\n    worst_signs = worst_5.index.tolist()\n    best_signs = best_5.index.tolist()\n    \n    print(\"\\nHypotheses for poor performance:\")\n    print(\"  1. Sample Size: Check if worst signs have fewer training examples\")\n    print(\"  2. Visual Similarity: Worst signs may be visually similar to others\")\n    print(\"  3. Motion Complexity: Short/simple signs may lack distinctive features\")\n    print(\"  4. Handshape Similarity: Similar handshapes between confused pairs\")\n    \n    # Check support (number of samples) for worst vs best\n    print(\"\\nSample Support Comparison:\")\n    print(f\"  Average support (worst 5): {worst_5['support'].mean():.1f}\")\n    print(f\"  Average support (best 5):  {best_5['support'].mean():.1f}\")\n    \n    # Recommendations\n    print(\"\\n\" + \"=\"*80)\n    print(\"RECOMMENDATIONS FOR IMPROVEMENT\")\n    print(\"=\"*80)\n    print(\"1. Data Augmentation:\")\n    print(\"   - Focus on worst-performing classes with temporal augmentations\")\n    print(\"   - Apply hand-specific augmentations for similar handshapes\")\n    print(\"\\n2. Model Architecture:\")\n    print(\"   - Consider attention mechanisms to focus on discriminative features\")\n    print(\"   - Add auxiliary losses for confused pairs\")\n    print(\"\\n3. Training Strategy:\")\n    print(\"   - Use class weights to handle imbalanced classes\")\n    print(\"   - Implement hard negative mining for confused pairs\")\n    print(\"\\n4. Feature Engineering:\")\n    print(\"   - Extract hand-shape specific features\")\n    print(\"   - Incorporate motion trajectory features\")\n    print(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:20.599119Z","iopub.execute_input":"2026-03-04T13:59:20.599345Z","iopub.status.idle":"2026-03-04T13:59:20.605972Z","shell.execute_reply.started":"2026-03-04T13:59:20.599323Z","shell.execute_reply":"2026-03-04T13:59:20.605285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Complete Classification Report (Optional)","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # This cell is optional - displays full report for all 250 classes\n    # Uncomment to see detailed metrics for every sign\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"COMPLETE CLASSIFICATION REPORT (All 250 Classes)\")\n    print(\"=\"*80)\n    print(\"Note: This is a comprehensive view. Scroll to see all classes.\")\n    print(\"=\"*80 + \"\\n\")\n    \n    full_report = class_metrics.sort_values('f1-score', ascending=False)\n    display(full_report[['precision', 'recall', 'f1-score', 'support']].round(4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:20.606777Z","iopub.execute_input":"2026-03-04T13:59:20.607087Z","iopub.status.idle":"2026-03-04T13:59:20.631557Z","shell.execute_reply.started":"2026-03-04T13:59:20.607059Z","shell.execute_reply":"2026-03-04T13:59:20.630939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history_metric(metric):\n    plt.figure(figsize=(15, 6))\n    plt.plot(history.history[metric], label='train')\n    if f'val_{metric}' in history.history:\n        plt.plot(history.history[f'val_{metric}'], label='val')\n    plt.title(f'Model {metric}')\n    plt.xlabel('Epoch')\n    plt.ylabel(metric)\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\nif TRAIN_MODEL:\n    plot_history_metric('loss')\n    plot_history_metric('acc')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T13:59:20.632302Z","iopub.execute_input":"2026-03-04T13:59:20.632523Z","iopub.status.idle":"2026-03-04T13:59:20.950579Z","shell.execute_reply.started":"2026-03-04T13:59:20.632494Z","shell.execute_reply":"2026-03-04T13:59:20.950007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_config(self):\n    \"\"\"Required for model serialization - allows model.save() to work\"\"\"\n    return {\n        'lr_max': float(self.lr_max),\n        'warmup_steps': int(self.warmup_steps),\n        'total_steps': int(self.total_steps),\n        'lr_min': float(self.lr_min)\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T14:08:08.188158Z","iopub.execute_input":"2026-03-04T14:08:08.188807Z","iopub.status.idle":"2026-03-04T14:08:08.192943Z","shell.execute_reply.started":"2026-03-04T14:08:08.188777Z","shell.execute_reply":"2026-03-04T14:08:08.192236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN_MODEL:\n    # Save model weights (always works)\n    model.save_weights('asl_transformer_weights.weights.h5')\n    print(\"Model weights saved successfully\")\n    \n    # Try to save full model\n    try:\n        model.save('asl_transformer_model.keras')\n        print(\"Full model saved successfully\")\n    except Exception as e:\n        print(\"Note: Full model save failed, but weights were saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T14:08:10.755835Z","iopub.execute_input":"2026-03-04T14:08:10.756469Z","iopub.status.idle":"2026-03-04T14:08:11.016688Z","shell.execute_reply.started":"2026-03-04T14:08:10.756440Z","shell.execute_reply":"2026-03-04T14:08:11.016048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rebuild model architecture (same code as training)\nmodel = get_model()\n\n# Load weights\nmodel.load_weights('asl_transformer_weights.weights.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T14:08:16.357766Z","iopub.execute_input":"2026-03-04T14:08:16.358394Z","iopub.status.idle":"2026-03-04T14:08:17.355321Z","shell.execute_reply.started":"2026-03-04T14:08:16.358364Z","shell.execute_reply":"2026-03-04T14:08:17.354605Z"}},"outputs":[],"execution_count":null}]}