{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"###  Environment Setup and Library Imports\n","metadata":{}},{"cell_type":"code","source":"# Standard System Libraries\nimport os\nimport sys\nimport gc\nimport time\nimport math\nimport random\nimport pickle\nimport glob\nimport datetime\n\n# Data Manipulation & Visualization\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\nfrom tqdm.autonotebook import tqdm\n\n# Machine Learning & Evaluation Metrics\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (accuracy_score, precision_score, \n                             recall_score, f1_score, \n                             classification_report, confusion_matrix)\n\n# Deep Learning (TensorFlow & Keras)\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model, load_model\nfrom tensorflow.keras.layers import (Dense, LSTM, Bidirectional, GRU, \n                                     Dropout, Input, LayerNormalization, \n                                     MultiHeadAttention, GlobalAveragePooling1D,\n                                     Conv1D, MaxPooling1D)\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras import mixed_precision\n\n# Configure random seeds for absolute reproducibility in research\nSEED = 42\nos.environ['PYTHONHASHSEED'] = str(SEED)\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\n# Print versions to document the research environment\nprint(\"--- Environment Details ---\")\nprint(f\"TensorFlow Version: {tf.__version__}\")\nprint(f\"Python Version: {sys.version.split()[0]}\")\nprint(f\"NumPy Version: {np.__version__}\")\nprint(f\"Pandas Version: {pd.__version__}\")\nprint(f\"Scikit-Learn Version: {sklearn.__version__}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:14.262756Z","iopub.execute_input":"2026-04-02T04:40:14.263108Z","iopub.status.idle":"2026-04-02T04:40:47.481358Z","shell.execute_reply.started":"2026-04-02T04:40:14.26307Z","shell.execute_reply":"2026-04-02T04:40:47.480551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport os\n\nprint(\"Current working directory:\", Path().resolve())\n\nDATA_DIR = Path(\"/kaggle/input/competitions/asl-signs\")\n\nTRAIN_CSV = DATA_DIR / \"train.csv\"\nLANDMARK_DIR = DATA_DIR / \"train_landmark_files\"\nSIGN_MAP = DATA_DIR / \"sign_to_prediction_index_map.json\"\n\nPROJECT_ROOT = Path(\"../\")\n\nEVAL_DIR = PROJECT_ROOT / \"Evaluation_Plots\"\nMODEL_DIR = PROJECT_ROOT / \"Saved_Models\"\nPRED_DIR = PROJECT_ROOT / \"Predictions\"\nHIST_DIR = PROJECT_ROOT / \"Training_Histories\"\n\nprint(\"\\nDataset paths check:\")\nprint(\"DATA_DIR:\", DATA_DIR)\nprint(\"Train CSV exists:\", TRAIN_CSV.exists())\nprint(\"Landmark folder exists:\", LANDMARK_DIR.exists())\nprint(\"Sign map exists:\", SIGN_MAP.exists())\n\nparquet_folders = list(LANDMARK_DIR.glob(\"*\"))\nprint(\"Number of parquet folders:\", len(parquet_folders))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:47.483098Z","iopub.execute_input":"2026-04-02T04:40:47.483625Z","iopub.status.idle":"2026-04-02T04:40:47.633454Z","shell.execute_reply.started":"2026-04-02T04:40:47.483594Z","shell.execute_reply":"2026-04-02T04:40:47.632468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(TRAIN_CSV)\ndisplay(train_df.head())\ndisplay(train_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:47.634514Z","iopub.execute_input":"2026-04-02T04:40:47.634779Z","iopub.status.idle":"2026-04-02T04:40:48.203531Z","shell.execute_reply.started":"2026-04-02T04:40:47.634752Z","shell.execute_reply":"2026-04-02T04:40:48.202813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of samples in train.csv:\", len(train_df))\nprint(\"Number of unique signs:\", train_df[\"sign\"].nunique())\nprint(\"Number of participants:\", train_df[\"participant_id\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:48.205372Z","iopub.execute_input":"2026-04-02T04:40:48.205642Z","iopub.status.idle":"2026-04-02T04:40:48.219855Z","shell.execute_reply.started":"2026-04-02T04:40:48.205616Z","shell.execute_reply":"2026-04-02T04:40:48.218939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Spatial-Temporal Feature Engineering and Preprocessing\nIn this phase, we define the core preprocessing pipeline. Instead of feeding all 543 raw MediaPipe landmarks into the models, we isolate the most informative nodes (Lips, Eyes, Nose, and Hands) to reduce noise and computational complexity. \n\nFurthermore, we implement a custom Keras Layer (`Preprocess`) that performs the following operations directly within the TensorFlow graph:\n1. **NaN Handling:** Computes safe means and standard deviations to normalize coordinates, replacing missing landmarks (NaNs) seamlessly.\n2. **Normalization:** Centers the coordinates based on a reference point.\n3. **Temporal Dynamics (Velocity & Acceleration):** Computes the first derivative (`dx`) and second derivative (`dx2`) of the coordinates across frames to capture motion speed and trajectory.\n4. **Feature Fusion:** Concatenates positions, velocities, and accelerations into a robust feature vector (shape: `[Frames, Channels]`) optimized for sequential models.","metadata":{}},{"cell_type":"code","source":"# Constants for data dimensions and padding\nROWS_PER_FRAME = 543\nMAX_LEN = 384\nCROP_LEN = MAX_LEN\nNUM_CLASSES  = 250\nPAD = -100.\n\n# ---------------------------------------------------------------------------\n# Feature Selection: Isolating Informative Landmarks (Lips, Nose, Eyes, Hands)\n# ---------------------------------------------------------------------------\nNOSE = [1, 2, 98, 327]\nLNOSE = [98]\nRNOSE = [327]\n\nLIP = [ \n    0, 61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\n\nLLIP = [84, 181, 91, 146, 61, 185, 40, 39, 37, 87, 178, 88, 95, 78, 191, 80, 81, 82]\nRLIP = [314, 405, 321, 375, 291, 409, 270, 269, 267, 317, 402, 318, 324, 308, 415, 310, 311, 312]\n\nPOSE = [500, 502, 504, 501, 503, 505, 512, 513]\nLPOSE = [513, 505, 503, 501]\nRPOSE = [512, 504, 502, 500]\n\nREYE = [\n    33, 7, 163, 144, 145, 153, 154, 155, 133,\n    246, 161, 160, 159, 158, 157, 173,\n]\nLEYE = [\n    263, 249, 390, 373, 374, 380, 381, 382, 362,\n    466, 388, 387, 386, 385, 384, 398,\n]\n\n# MediaPipe Hand Landmarks indices\nLHAND = np.arange(468, 489).tolist()\nRHAND = np.arange(522, 543).tolist()\n\n# Final concatenated feature list\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE\n\nNUM_NODES = len(POINT_LANDMARKS)\n# Channels = (X, Y) * (Position, Velocity, Acceleration) = 2 * 3 = 6 per node\nCHANNELS = 6 * NUM_NODES \n\nprint(f\"Total Selected Nodes: {NUM_NODES}\")\nprint(f\"Total Output Channels per frame: {CHANNELS}\")\n\n# ---------------------------------------------------------------------------\n# Utility Functions for robust mathematical operations\n# ---------------------------------------------------------------------------\ndef tf_nan_mean(x, axis=0, keepdims=False):\n    \"\"\"Computes the mean of a tensor ignoring NaN values.\"\"\"\n    sum_val = tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis, keepdims=keepdims)\n    count_val = tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis, keepdims=keepdims)\n    return sum_val / count_val\n\ndef tf_nan_std(x, center=None, axis=0, keepdims=False):\n    \"\"\"Computes the standard deviation of a tensor ignoring NaN values.\"\"\"\n    if center is None:\n        center = tf_nan_mean(x, axis=axis,  keepdims=True)\n    d = x - center\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis, keepdims=keepdims))\n\n# ---------------------------------------------------------------------------\n# Custom Keras Layer for Spatial-Temporal Feature Engineering\n# ---------------------------------------------------------------------------\nclass Preprocess(tf.keras.layers.Layer):\n    \"\"\"\n    A custom TensorFlow layer that normalizes coordinates, handles NaNs, \n    and computes dynamic temporal features (velocity and acceleration).\n    \"\"\"\n    def __init__(self, max_len=MAX_LEN, point_landmarks=POINT_LANDMARKS, **kwargs):\n        super().__init__(**kwargs)\n        self.max_len = max_len\n        self.point_landmarks = point_landmarks\n\n    def call(self, inputs):\n        if  inputs.shape.rank == 3:\n            x = inputs[None, ...]\n        else:\n            x = inputs\n        \n        # Center normalization based on a reference point\n        mean = tf_nan_mean(tf.gather(x, [17], axis=2), axis=[1, 2], keepdims=True)\n        mean = tf.where(tf.math.is_nan(mean), tf.constant(0.5, x.dtype), mean)\n        \n        # Isolate selected landmarks\n        x = tf.gather(x, self.point_landmarks, axis=2) # Shape: N, T, P, C\n        std = tf_nan_std(x, center=mean, axis=[1, 2], keepdims=True)\n        x = (x - mean) / std\n\n        if self.max_len is not None:\n            x = x[:, :self.max_len]\n            \n        length = tf.shape(x)[1]\n        \n        # Retain only X and Y coordinates (drop Z for classification efficiency)\n        x = x[..., :2]\n\n        # Calculate Velocity (First Derivative - dx)\n        dx = tf.cond(\n            tf.shape(x)[1] > 1,\n            lambda: tf.pad(x[:, 1:] - x[:, :-1], [[0, 0], [0, 1], [0, 0], [0, 0]]),\n            lambda: tf.zeros_like(x)\n        )\n\n        # Calculate Acceleration (Second Derivative - dx2)\n        dx2 = tf.cond(\n            tf.shape(x)[1] > 2,\n            lambda: tf.pad(x[:, 2:] - x[:, :-2], [[0, 0], [0, 2], [0, 0], [0, 0]]),\n            lambda: tf.zeros_like(x)\n        )\n\n        # Concatenate Position, Velocity, and Acceleration\n        x = tf.concat([\n            tf.reshape(x, (-1, length, 2 * len(self.point_landmarks))),\n            tf.reshape(dx, (-1, length, 2 * len(self.point_landmarks))),\n            tf.reshape(dx2, (-1, length, 2 * len(self.point_landmarks))),\n        ], axis=-1)\n        \n        # Replace any remaining NaNs with zeros\n        x = tf.where(tf.math.is_nan(x), tf.constant(0., x.dtype), x)\n        \n        return x\n\n    def get_config(self):\n        \"\"\"Required for layer serialization and model saving.\"\"\"\n        config = super().get_config()\n        config.update({\n            \"max_len\": self.max_len,\n            \"point_landmarks\": self.point_landmarks,\n        })\n        return config","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:48.221159Z","iopub.execute_input":"2026-04-02T04:40:48.221512Z","iopub.status.idle":"2026-04-02T04:40:48.244718Z","shell.execute_reply.started":"2026-04-02T04:40:48.221471Z","shell.execute_reply":"2026-04-02T04:40:48.243979Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Phase 4: Data Augmentation and Parquet Pipeline Integration\nThis section adapts the standard TFRecord-based data loading pipeline to directly read from `.parquet` files using a Python generator wrapped in `tf.data.Dataset.from_generator`. \n\nIt includes advanced spatial-temporal augmentations specifically designed for sign language recognition:\n1. `flip_lr`: Simulates left-handed vs right-handed signers.\n2. `resample`: Alters the speed of the sign dynamically.\n3. `spatial_random_affine`: Applies rotation, scaling, and shear to simulate different camera angles.\n4. `spatial_mask` & `temporal_mask`: Adds robustness by randomly obscuring parts of the frame or sequence.\n\nFinally, the `get_parquet_dataset` function constructs an optimized, prefetching TensorFlow dataset ready for model training.","metadata":{}},{"cell_type":"code","source":"# 1. Encode Sign Labels to Integers (0 to 249)\nif 'label' not in train_df.columns:\n    sign_list = sorted(train_df['sign'].unique())\n    sign_to_label = {sign: label for label, sign in enumerate(sign_list)}\n    label_to_sign = {label: sign for sign, label in sign_to_label.items()}\n    train_df['label'] = train_df['sign'].map(sign_to_label)\n    print(f\"Encoded {len(sign_list)} unique signs.\")\n\n# Initialize the Preprocess layer defined in the previous cell\npreprocess_layer = Preprocess(max_len=MAX_LEN, point_landmarks=POINT_LANDMARKS)\n\n# ---------------------------------------------------------------------------\n# Core Parquet Reader Function\n# ---------------------------------------------------------------------------\ndef load_parquet_video(file_path):\n    try:\n        df = pd.read_parquet(file_path, columns=['x', 'y', 'z'], engine='pyarrow')\n        coords = df.values.astype(np.float32)\n        frames = len(coords) // ROWS_PER_FRAME\n        return coords.reshape(frames, ROWS_PER_FRAME, 3)\n    except Exception as e:\n        return np.zeros((0, ROWS_PER_FRAME, 3), dtype=np.float32)\n\n# ---------------------------------------------------------------------------\n# Data Augmentation Functions (Corrected Tensor Dimensions)\n# ---------------------------------------------------------------------------\ndef filter_nans_tf(x, ref_point=POINT_LANDMARKS):\n    mask = tf.math.logical_not(tf.reduce_all(tf.math.is_nan(tf.gather(x, ref_point, axis=1)), axis=[-2, -1]))\n    x = tf.boolean_mask(x, mask, axis=0)\n    return x\n\ndef flip_lr(x):\n    x_coord, y_coord, z_coord = tf.unstack(x, axis=-1)\n    x_coord = 1 - x_coord\n    new_x = tf.stack([x_coord, y_coord, z_coord], -1)\n    new_x = tf.transpose(new_x, [1, 0, 2])\n\n    lhand = tf.gather(new_x, LHAND, axis=0)\n    rhand = tf.gather(new_x, RHAND, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LHAND)[..., None], rhand)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RHAND)[..., None], lhand)\n\n    llip = tf.gather(new_x, LLIP, axis=0)\n    rlip = tf.gather(new_x, RLIP, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LLIP)[..., None], rlip)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RLIP)[..., None], llip)\n\n    lpose = tf.gather(new_x, LPOSE, axis=0)\n    rpose = tf.gather(new_x, RPOSE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LPOSE)[..., None], rpose)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RPOSE)[..., None], lpose)\n\n    leye = tf.gather(new_x, LEYE, axis=0)\n    reye = tf.gather(new_x, REYE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LEYE)[..., None], reye)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(REYE)[..., None], leye)\n\n    lnose = tf.gather(new_x, LNOSE, axis=0)\n    rnose = tf.gather(new_x, RNOSE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LNOSE)[..., None], rnose)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RNOSE)[..., None], lnose)\n\n    new_x = tf.transpose(new_x, [1, 0, 2])\n    return new_x\n\ndef interp1d_(x, target_len, method='random'):\n    target_len = tf.maximum(1, target_len)\n    width = tf.shape(x)[1]\n    size = [target_len, width]\n\n    if method == 'random':\n        rand_val = tf.random.uniform(())\n        if rand_val < 0.33:\n            x = tf.image.resize(x, size, 'bilinear')\n        elif rand_val < 0.66:\n            x = tf.image.resize(x, size, 'bicubic')\n        else:\n            x = tf.image.resize(x, size, 'nearest')\n    else:\n        x = tf.image.resize(x, size, method)\n    return x\n\ndef resample(x, rate=(0.8, 1.2)):\n    rate = tf.random.uniform((), rate[0], rate[1])\n    length = tf.shape(x)[0]\n    new_size = tf.cast(rate * tf.cast(length, tf.float32), tf.int32)\n    new_x = interp1d_(x, new_size)\n    return new_x\n\ndef spatial_random_affine(xyz, scale=(0.8, 1.2), shear=(-0.15, 0.15), shift=(-0.1, 0.1), degree=(-30, 30)):\n    center = tf.constant([0.5, 0.5])\n    if scale is not None:\n        scale_val = tf.random.uniform((), *scale)\n        xyz = scale_val * xyz\n\n    if shear is not None:\n        xy = xyz[..., :2]\n        z = xyz[..., 2:]\n        shear_x = shear_y = tf.random.uniform((), *shear)\n        if tf.random.uniform(()) < 0.5:\n            shear_x = 0.\n        else:\n            shear_y = 0.\n        shear_mat = tf.identity([[1., shear_x], [shear_y, 1.]])\n        xy = xy @ shear_mat\n        center = center + [shear_y, shear_x]\n        xyz = tf.concat([xy, z], axis=-1)\n\n    if degree is not None:\n        xy = xyz[..., :2]\n        z = xyz[..., 2:]\n        xy -= center\n        degree_val = tf.random.uniform((), *degree)\n        radian = degree_val / 180 * np.pi\n        c = tf.math.cos(radian)\n        s = tf.math.sin(radian)\n        rotate_mat = tf.identity([[c, s], [-s, c]])\n        xy = xy @ rotate_mat\n        xy = xy + center\n        xyz = tf.concat([xy, z], axis=-1)\n\n    if shift is not None:\n        shift_val = tf.random.uniform((), *shift)\n        xyz = xyz + shift_val\n\n    return xyz\n\ndef temporal_crop(x, length=MAX_LEN):\n    l = tf.shape(x)[0]\n    offset = tf.random.uniform((), 0, tf.clip_by_value(l - length, 1, length), dtype=tf.int32)\n    x = x[offset:offset + length]\n    return x\n\ndef temporal_mask(x, size=(0.2, 0.4), mask_value=float('nan')):\n    l = tf.shape(x)[0]\n    mask_size = tf.random.uniform((), *size)\n    mask_size = tf.cast(tf.cast(l, tf.float32) * mask_size, tf.int32)\n    mask_offset = tf.random.uniform((), 0, tf.clip_by_value(l - mask_size, 1, l), dtype=tf.int32)\n    indices = tf.range(mask_offset, mask_offset + mask_size)[..., None]\n    updates = tf.fill([mask_size, ROWS_PER_FRAME, 3], mask_value)\n    x = tf.tensor_scatter_nd_update(x, indices, updates)\n    return x\n\ndef spatial_mask(x, size=(0.2, 0.4), mask_value=float('nan')):\n    mask_offset_y = tf.random.uniform(())\n    mask_offset_x = tf.random.uniform(())\n    mask_size = tf.random.uniform((), *size)\n    mask_x = (mask_offset_x < x[..., 0]) & (x[..., 0] < mask_offset_x + mask_size)\n    mask_y = (mask_offset_y < x[..., 1]) & (x[..., 1] < mask_offset_y + mask_size)\n    mask = mask_x & mask_y\n    x = tf.where(mask[..., None], mask_value, x)\n    return x\n\ndef augment_fn(x, max_len=None):\n    if tf.random.uniform(()) < 0.8:\n        x = resample(x, (0.5, 1.5))\n    if tf.random.uniform(()) < 0.5:\n        x = flip_lr(x)\n    if max_len is not None:\n        x = temporal_crop(x, max_len)\n    if tf.random.uniform(()) < 0.75:\n        x = spatial_random_affine(x)\n    if tf.random.uniform(()) < 0.5:\n        x = temporal_mask(x)\n    if tf.random.uniform(()) < 0.5:\n        x = spatial_mask(x)\n    return x\n\n# ---------------------------------------------------------------------------\n# TensorFlow Data Pipeline Implementation\n# ---------------------------------------------------------------------------\ndef process_data(coord, label, augment=False, max_len=MAX_LEN):\n    coord = filter_nans_tf(coord)\n    if augment:\n        coord = augment_fn(coord, max_len=max_len)\n    coord = tf.ensure_shape(coord, (None, ROWS_PER_FRAME, 3))\n\n    processed = preprocess_layer(coord)\n    processed = tf.squeeze(processed, axis=0)\n    processed = tf.cast(processed, tf.float32)\n\n    one_hot_label = tf.one_hot(label, NUM_CLASSES)\n    return processed, one_hot_label\n\ndef get_parquet_dataset(df, data_dir=DATA_DIR, batch_size=64, max_len=MAX_LEN, augment=False, shuffle=False):\n    def generator():\n        sample_df = df.sample(frac=1).reset_index(drop=True) if shuffle else df\n        for _, row in sample_df.iterrows():\n            file_path = os.path.join(data_dir, str(row['path']).replace('\\\\', '/'))\n            file_path = os.path.normpath(file_path)\n\n            coords = load_parquet_video(file_path)\n            label = int(row['label'])\n            if coords.shape[0] > 0:\n                yield coords, label\n\n    ds = tf.data.Dataset.from_generator(\n        generator,\n        output_signature=(\n            tf.TensorSpec(shape=(None, ROWS_PER_FRAME, 3), dtype=tf.float32),\n            tf.TensorSpec(shape=(), dtype=tf.int32)\n        )\n    )\n\n    ds = ds.map(lambda x, y: process_data(x, y, augment=augment, max_len=max_len),\n                num_parallel_calls=tf.data.AUTOTUNE)\n\n    ds = ds.padded_batch(\n        batch_size,\n        padding_values=(tf.cast(PAD, tf.float32), tf.cast(0.0, tf.float32)),\n        padded_shapes=([max_len, CHANNELS], [NUM_CLASSES]),\n        drop_remainder=True\n    )\n\n    ds = ds.repeat()          # ADDED: prevents dataset exhaustion across epochs\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    return ds\n\n# ---------------------------------------------------------------------------\n# Pipeline Sanity Check\n# ---------------------------------------------------------------------------\nprint(\"Testing the Parquet Pipeline...\")\ntest_df_subset = train_df.head(10)\ntest_ds = get_parquet_dataset(test_df_subset, batch_size=2, augment=True)\n\nfor batch_x, batch_y in test_ds.take(1):\n    print(f\"Batch X Shape: {batch_x.shape}\")\n    print(f\"Batch Y Shape: {batch_y.shape}\")\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:48.245835Z","iopub.execute_input":"2026-04-02T04:40:48.246177Z","iopub.status.idle":"2026-04-02T04:40:52.279581Z","shell.execute_reply.started":"2026-04-02T04:40:48.24615Z","shell.execute_reply":"2026-04-02T04:40:52.278575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  Visualizing MediaPipe Landmarks \nBefore feeding the data into complex neural networks, it is essential to visually verify the integrity of the spatial coordinates. The following code iterates through the parquet files, locates a sequence with valid hand landmarks (filtering out missing frames), and generates an interactive 2D animation of the hand skeletal connections over time. This confirms that the coordinate extraction and reshaping processes are correct.","metadata":{}},{"cell_type":"code","source":"from IPython.display import HTML\nimport matplotlib.pyplot as plt\nfrom matplotlib.animation import FuncAnimation\nimport numpy as np\nimport os\n\"\"\"\n\n# ---------------------------------------------------------\n# MediaPipe Hand Connections\n# Defines skeletal edges between the 21 hand landmarks\n# ---------------------------------------------------------\n\nHAND_EDGES = [\n    (0,1),(1,2),(2,3),(3,4),\n    (0,5),(5,6),(6,7),(7,8),\n    (5,9),(9,10),(10,11),(11,12),\n    (9,13),(13,14),(14,15),(15,16),\n    (13,17),(0,17),(17,18),(18,19),(19,20)\n]\n\n\n# ---------------------------------------------------------\n# Utility: Filter frames that contain only NaN coordinates\n# Some sequences contain missing frames that should be removed\n# ---------------------------------------------------------\n\ndef filter_nans(frames):\n    mask = ~np.isnan(frames).all(axis=(-2,-1))\n    return frames[mask]\n\n\n# ---------------------------------------------------------\n# Locate a valid sequence in the dataset\n# The sequence must contain a visible hand motion\n# ---------------------------------------------------------\n\nsample_frames = None\n\nprint(\"Searching for a valid sequence\")\n\nfor row in train_df.itertuples():\n\n    file_path = os.path.join(DATA_DIR, str(row.path).replace(\"\\\\\",\"/\"))\n    file_path = os.path.normpath(file_path)\n\n    coords = load_parquet_video(file_path)\n\n    if coords.shape[0] == 0:\n        continue\n\n    lhand = coords[:,LHAND,:]\n\n    valid_frames = filter_nans(lhand)\n\n    if len(valid_frames) > 20:\n        sample_frames = coords\n        print(\"Sequence found:\", row.sign)\n        break\n\n\n# ---------------------------------------------------------\n# Core Animation Function\n# Draws landmarks and skeleton edges frame-by-frame\n# ---------------------------------------------------------\n\ndef animate_frames(frames, edges=None, idxs=None):\n\n    frames = filter_nans(frames)\n\n    fig, ax = plt.subplots(figsize=(6,6))\n\n    def plot_frame(i):\n\n        ax.clear()\n\n        frame = np.nan_to_num(frames[i])\n\n        x = frame[:,0]\n        y = frame[:,1]\n\n        ax.scatter(x,y,color=\"dodgerblue\",s=40)\n\n        if idxs is not None:\n            for j in range(len(x)):\n                ax.text(x[j],y[j],str(idxs[j]),fontsize=7)\n\n        if edges is not None:\n            for e in edges:\n                ax.plot(\n                    [x[e[0]],x[e[1]]],\n                    [y[e[0]],y[e[1]]],\n                    color=\"salmon\",\n                    linewidth=2\n                )\n\n        ax.invert_yaxis()\n\n        ax.set_xticks([])\n        ax.set_yticks([])\n\n    anim = FuncAnimation(fig, plot_frame, frames=len(frames), interval=100)\n\n    plt.close(fig)\n\n    return HTML(anim.to_jshtml())\n\n\n# ---------------------------------------------------------\n# Save Animation Function\n# Used to export animations for research paper figures\n# ---------------------------------------------------------\n\ndef save_animation(frames, filename, edges=None, idxs=None):\n\n    frames = filter_nans(frames)\n\n    fig, ax = plt.subplots(figsize=(6,6))\n\n    def plot_frame(i):\n\n        ax.clear()\n\n        frame = np.nan_to_num(frames[i])\n\n        x = frame[:,0]\n        y = frame[:,1]\n\n        ax.scatter(x,y,color=\"dodgerblue\",s=40)\n\n        if idxs is not None:\n            for j in range(len(x)):\n                ax.text(x[j],y[j],str(idxs[j]),fontsize=7)\n\n        if edges is not None:\n            for e in edges:\n                ax.plot([x[e[0]],x[e[1]]],[y[e[0]],y[e[1]]],color=\"salmon\")\n\n        ax.invert_yaxis()\n\n        ax.set_xticks([])\n        ax.set_yticks([])\n\n    anim = FuncAnimation(fig, plot_frame, frames=len(frames), interval=100)\n\n    save_path = os.path.join(\"Research Paper\",\"Evaluation_Plots\",filename)\n\n    anim.save(save_path, writer=\"pillow\", fps=10)\n\n    plt.close(fig)\n\n    print(\"Animation saved to:\", save_path)\n\n\n# ---------------------------------------------------------\n# Display Left Hand Landmarks\n# ---------------------------------------------------------\n\nprint(\"Left Hand Motion\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,LHAND],\n        edges=HAND_EDGES,\n        idxs=list(range(len(LHAND)))\n    )\n)\n\n\n# ---------------------------------------------------------\n# Display Right Hand Landmarks\n# ---------------------------------------------------------\n\nprint(\"Right Hand Motion\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,RHAND],\n        edges=HAND_EDGES,\n        idxs=list(range(len(RHAND)))\n    )\n)\n\n\n# ---------------------------------------------------------\n# Display Face Landmarks\n# ---------------------------------------------------------\n\nprint(\"Face Landmarks\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,LIP + LEYE + REYE + NOSE],\n        idxs=LIP + LEYE + REYE + NOSE\n    )\n)\n\n\n# ---------------------------------------------------------\n# Display All Selected Landmarks Used by the Model\n# ---------------------------------------------------------\n\nprint(\"Full Landmark Representation\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,POINT_LANDMARKS],\n        idxs=POINT_LANDMARKS\n    )\n)\n\n\n# ---------------------------------------------------------\n# Example of Augmented Sequence Visualization\n# ---------------------------------------------------------\n\nprint(\"Augmented Sequence Example\")\n\naugmented = augment_fn(sample_frames, max_len=MAX_LEN).numpy()\n\ndisplay(\n    animate_frames(\n        augmented[:,POINT_LANDMARKS],\n        idxs=POINT_LANDMARKS\n    )\n)\n\n\n# ---------------------------------------------------------\n# Save animations for later use\n# ---------------------------------------------------------\n\nsave_animation(sample_frames[:,RHAND],\"right_hand.gif\",edges=HAND_EDGES,idxs=list(range(len(RHAND))))\nsave_animation(sample_frames[:,POINT_LANDMARKS],\"full_landmarks.gif\",idxs=POINT_LANDMARKS)\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:52.280835Z","iopub.execute_input":"2026-04-02T04:40:52.281229Z","iopub.status.idle":"2026-04-02T04:40:52.308811Z","shell.execute_reply.started":"2026-04-02T04:40:52.281187Z","shell.execute_reply":"2026-04-02T04:40:52.307778Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  Final Data Shape and Tensor Inspection\nBefore defining the neural network architectures, we extract a single batch from our parquet dataset generator to inspect the exact tensor dimensions and the statistical properties of the engineered features. This confirms that the normalization, padding, and one-hot encoding have been applied correctly.","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\nprint(\"Fetching a single batch from the dataset...\")\n# We use the test_ds created in the previous cell\nfor batch_x, batch_y in test_ds.take(1):\n    \n    x_numpy = batch_x.numpy()\n    y_numpy = batch_y.numpy()\n    \n    print(\"\\n--- 1. Input Features (X) ---\")\n    print(f\"Shape: {x_numpy.shape} -> (Batch, Frames, Channels)\")\n    print(f\"Data Type: {x_numpy.dtype}\")\n    # Check normalization properties (should be centered around 0)\n    print(f\"Global Min Value: {np.min(x_numpy):.4f}\")\n    print(f\"Global Max Value: {np.max(x_numpy):.4f}\")\n    print(f\"Global Mean: {np.mean(x_numpy):.4f}\")\n    \n    print(\"\\n--- 2. Output Labels (Y) ---\")\n    print(f\"Shape: {y_numpy.shape} -> (Batch, Num_Classes)\")\n    print(f\"Data Type: {y_numpy.dtype}\")\n    \n    # Show the active class for the first sample in the batch\n    first_sample_label_index = np.argmax(y_numpy[0])\n    # Assuming label_to_sign dictionary is available from previous cell\n    sign_word = label_to_sign.get(first_sample_label_index, \"Unknown\")\n    print(f\"First Sample One-Hot Label Index: {first_sample_label_index}\")\n    print(f\"Corresponding Sign Word: '{sign_word}'\")\n    \n    print(\"\\n Data is perfectly shaped and ready for modeling!\")\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:52.310218Z","iopub.execute_input":"2026-04-02T04:40:52.310723Z","iopub.status.idle":"2026-04-02T04:40:52.63303Z","shell.execute_reply.started":"2026-04-02T04:40:52.310693Z","shell.execute_reply":"2026-04-02T04:40:52.631773Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Phase 2: Stratified Data Splitting and Dataset Creation\nTo ensure robust model evaluation and prevent data leakage, we split the dataset into Training (80%), Validation (10%), and Testing (10%) sets. \n\nKey considerations for this research pipeline:\n1. **Stratification:** We stratify the split based on the target labels to maintain a consistent class distribution across all splits (crucial for the 250-class ISLR dataset).\n2. **Reproducibility:** A fixed random seed ensures identical splits across different execution environments.\n3. **Artifact Saving:** The splits are saved as CSV files. This guarantees that all subsequent baseline and advanced models, as well as the final Ensemble, are evaluated on the exact same unseen test instances.\n4. **Optimized Pipelines:** We instantiate `tf.data` pipelines with `AUTOTUNE` prefetching. Augmentation and shuffling are strictly applied only to the training set.","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\n\n\nprint(\"Initiating Stratified Data Splitting...\")\n\n# 1. Create data directory if it doesn't exist (from your project structure)\nos.makedirs(\"data\", exist_ok=True)\n\n\ntrain_df_split, temp_df = train_test_split(\n    train_df, \n    test_size=0.20, \n    random_state=SEED, \n    stratify=train_df['label']\n)\n\n# Second split: Split the 40% Temporary equally into 20% Validation and 20% Test\nval_df_split, test_df_split = train_test_split(\n    temp_df, \n    test_size=0.50, \n    random_state=SEED, \n    stratify=temp_df['label']\n)\n\n# 3. Save the splits to disk for absolute reproducibility during Ensemble\ntrain_split_path = os.path.join(\"data\", \"train_split.csv\")\nval_split_path = os.path.join(\"data\", \"val_split.csv\")\ntest_split_path = os.path.join(\"data\", \"test_split.csv\")\n\ntrain_df_split.to_csv(train_split_path, index=False)\nval_df_split.to_csv(val_split_path, index=False)\ntest_df_split.to_csv(test_split_path, index=False)\n\nprint(f\"Data Splitting Complete and Saved to 'data/' directory.\")\nprint(f\"Total Samples: {len(train_df)}\")\nprint(f\"--> Training Set:   {len(train_df_split)} samples ({len(train_df_split)/len(train_df)*100:.1f}%)\")\nprint(f\"--> Validation Set: {len(val_df_split)} samples ({len(val_df_split)/len(train_df)*100:.1f}%)\")\nprint(f\"--> Testing Set:    {len(test_df_split)} samples ({len(test_df_split)/len(train_df)*100:.1f}%)\")\n\n# 4. Create Highly Optimized tf.data.Datasets\nprint(\"\\nConstructing TensorFlow Datasets...\")\n\n# Hyperparameters for training\nBATCH_SIZE = 128 # Adjust this depending on your GPU RAM (e.g., 32 if OOM error occurs, 128 if plenty of VRAM)\n\n# Train Dataset: Needs Augmentation and Shuffling\ntrain_dataset = get_parquet_dataset(\n    train_df_split, \n    data_dir=DATA_DIR, \n    batch_size=BATCH_SIZE, \n    max_len=MAX_LEN, \n    augment=True, \n    shuffle=True\n)\n\n# Validation Dataset: NO Augmentation, NO Shuffling (for accurate metric tracking)\nval_dataset = get_parquet_dataset(\n    val_df_split, \n    data_dir=DATA_DIR, \n    batch_size=BATCH_SIZE, \n    max_len=MAX_LEN, \n    augment=False, \n    shuffle=False\n)\n\n# Test Dataset: NO Augmentation, NO Shuffling (for final paper evaluation)\ntest_dataset = get_parquet_dataset(\n    test_df_split, \n    data_dir=DATA_DIR, \n    batch_size=BATCH_SIZE, \n    max_len=MAX_LEN, \n    augment=False, \n    shuffle=False\n)\n\nprint(\"\\n--- TF Dataset Specifications ---\")\nprint(f\"Train Dataset: {train_dataset}\")\nprint(f\"Val Dataset:   {val_dataset}\")\nprint(f\"Test Dataset:  {test_dataset}\")\nprint(\" Data Pipelines are heavily optimized and ready for model consumption!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:52.63427Z","iopub.execute_input":"2026-04-02T04:40:52.634639Z","iopub.status.idle":"2026-04-02T04:40:53.986391Z","shell.execute_reply.started":"2026-04-02T04:40:52.6346Z","shell.execute_reply":"2026-04-02T04:40:53.985403Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dataset Summary and Pre-Training Report\nBefore initializing the training phase, we generate a comprehensive statistical report of the engineered dataset. This step verifies the integrity of the stratified split and the exact tensor dimensions. A textual summary is exported to the `results` directory, and a visual representation of the data distribution is saved to the `Evaluation_Plots` directory for inclusion in the research methodology section.","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/working/Wessal_Project\"\n\nFOLDERS = [\n    \"architecture_model\",\n    \"data\",\n    \"Evaluation_Plots\",\n    \"logs\",\n    \"notebooks\",\n    \"Predictions\",\n    \"results\",\n    \"Saved_Models\",\n    \"Training_Histories\"\n]\n\nfor folder in FOLDERS:\n    os.makedirs(os.path.join(BASE_DIR, folder), exist_ok=True)\n\nprint(\"Project structure created successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:53.989296Z","iopub.execute_input":"2026-04-02T04:40:53.989579Z","iopub.status.idle":"2026-04-02T04:40:53.995957Z","shell.execute_reply.started":"2026-04-02T04:40:53.989552Z","shell.execute_reply":"2026-04-02T04:40:53.995137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# 1. Compile the Statistical Data\ntotal_samples = len(train_df)\ntrain_samples = len(train_df_split)\nval_samples = len(val_df_split)\ntest_samples = len(test_df_split)\n\ntrain_class_counts = train_df_split['label'].value_counts()\nval_class_counts = val_df_split['label'].value_counts()\ntest_class_counts = test_df_split['label'].value_counts()\n\n# 2. Generate the Textual Report\nreport_text = f\"\"\"\n=========================================================\n          WESSAL PROJECT: DATASET SUMMARY REPORT\n=========================================================\n1. GLOBAL DATASET METRICS\n---------------------------------------------------------\nTotal Video Sequences Analyzed : {total_samples}\nTotal Unique Sign Classes      : {NUM_CLASSES}\nSelected Landmarks per Frame   : {NUM_NODES} nodes\nEngineered Feature Channels    : {CHANNELS} (X, Y, dx, dy, dx2, dy2)\nMaximum Sequence Length        : {MAX_LEN} frames\nFinal Input Tensor Shape       : (Batch_Size, {MAX_LEN}, {CHANNELS})\n\n2. STRATIFIED DATA SPLITTING (80/10/10)\n---------------------------------------------------------\nTraining Set (80%)             : {train_samples} samples\nValidation Set (10%)           : {val_samples} samples\nTesting Set (10%)              : {test_samples} samples\n\n3. CLASS BALANCE VERIFICATION (Samples per Class)\n---------------------------------------------------------\n[Training Set]   Max: {train_class_counts.max()} | Min: {train_class_counts.min()} | Mean: {train_class_counts.mean():.1f}\n[Validation Set] Max: {val_class_counts.max()}  | Min: {val_class_counts.min()}  | Mean: {val_class_counts.mean():.1f}\n[Testing Set]    Max: {test_class_counts.max()}  | Min: {test_class_counts.min()}  | Mean: {test_class_counts.mean():.1f}\n=========================================================\n\"\"\"\n\n# Print to console\nprint(report_text)\n\n# Save report to text file\nreport_path = os.path.join(BASE_DIR, \"Pre_Training_Dataset_Report.txt\")\nwith open(report_path, \"w\") as text_file:\n    text_file.write(report_text)\nprint(f\"Report successfully saved to: {report_path}\")\n\n# 3. Generate Visual Report (Plots)\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 6))\n\n# Subplot 1: Pie Chart of the Split\nlabels = ['Training (60%)', 'Validation (20%)', 'Testing (20%)']\nsizes = [train_samples, val_samples, test_samples]\ncolors = ['#4285F4', '#34A853', '#FBBC05']\nexplode = (0.05, 0, 0)  \n\nax1.pie(sizes, explode=explode, labels=labels, colors=colors, autopct='%1.1f%%',\n        shadow=False, startangle=90, textprops={'fontsize': 12})\nax1.axis('equal') \nax1.set_title('Dataset Allocation (Stratified Split)', fontsize=14, fontweight='bold', pad=15)\n\n# Subplot 2: Bar Chart showing Class Balance (Mean samples per class)\nsplit_names = ['Train', 'Validation', 'Test']\nmean_samples = [train_class_counts.mean(), val_class_counts.mean(), test_class_counts.mean()]\n\nax2.bar(split_names, mean_samples, color=['#4285F4', '#34A853', '#FBBC05'], width=0.5)\nax2.set_ylabel('Mean Sequences per Class', fontsize=12)\nax2.set_title('Average Class Representation per Split', fontsize=14, fontweight='bold', pad=15)\n\nfor i, v in enumerate(mean_samples):\n    ax2.text(i, v + 2, f\"{v:.1f}\", ha='center', va='bottom', fontsize=11, fontweight='bold')\n\nplt.tight_layout()\n\n# Save the plot\nplot_path = os.path.join(BASE_DIR, \"Data_Split_Distribution.png\")\nplt.savefig(plot_path, dpi=300, bbox_inches='tight')\nprint(f\"Visual distribution plot saved to: {plot_path}\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:53.997217Z","iopub.execute_input":"2026-04-02T04:40:53.997642Z","iopub.status.idle":"2026-04-02T04:40:55.072361Z","shell.execute_reply.started":"2026-04-02T04:40:53.997613Z","shell.execute_reply":"2026-04-02T04:40:55.071548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\n# 1. Custom Conv1D to explicitly preserve the temporal mask\n\nclass MaskableConv1D(layers.Conv1D):\n    def __init__(self, *args, **kwargs):\n        super().__init__(*args, **kwargs)\n        self.supports_masking = True\n\ndef get_model(max_len, channels, num_classes, dim=192):\n    inp = layers.Input(shape=(max_len, channels), name='input_features')\n\n    # -----------------------------------------------------------------------\n    # Native Masking & Explicit Attention Mask (Fixed with Lambda Layer)\n    # -----------------------------------------------------------------------\n    # Automatic mask for GRU\n    x = layers.Masking(mask_value=0.0, name='masking_layer')(inp)\n    \n    # Wrap the raw TF operations in a Lambda layer to comply with Keras requirements\n    def generate_attention_mask(inputs):\n        bool_mask = tf.reduce_any(tf.not_equal(inputs, 0.0), axis=-1)\n        return bool_mask[:, tf.newaxis, :]\n        \n    attn_mask = layers.Lambda(generate_attention_mask, name='attn_mask_generator')(inp)\n\n    # -----------------------------------------------------------------------\n    # Stem\n    # -----------------------------------------------------------------------\n    x = layers.Dense(dim, use_bias=False, name='stem_dense')(x)\n    x = layers.LayerNormalization(epsilon=1e-6, name='stem_ln')(x)\n    x = layers.Activation('swish')(x)\n\n    # -----------------------------------------------------------------------\n    # Stage 1: Conv1D (Using MaskableConv1D to pass the mask safely)\n    # -----------------------------------------------------------------------\n    for i in range(3):\n        residual = x\n        x = layers.LayerNormalization(epsilon=1e-6)(x)\n        x = MaskableConv1D(dim, 17, padding='same', activation='swish', use_bias=False)(x)\n        x = layers.SpatialDropout1D(0.1)(x)\n        x = layers.Add()([residual, x])\n\n    # -----------------------------------------------------------------------\n    # Stage 2: Transformer (With Explicit Attention Mask)\n    # -----------------------------------------------------------------------\n    for i in range(2):\n        # Attention Branch\n        residual = x\n        x = layers.LayerNormalization(epsilon=1e-6)(x)\n        attn_out = layers.MultiHeadAttention(num_heads=4, key_dim=dim//4, dropout=0.1)(\n            query=x, value=x, attention_mask=attn_mask\n        )\n        x = layers.Add()([residual, attn_out])\n\n        # Feed-Forward Branch\n        residual = x\n        x = layers.LayerNormalization(epsilon=1e-6)(x)\n        x = layers.Dense(dim * 2, activation='swish', use_bias=False)(x)\n        x = layers.Dense(dim, use_bias=False)(x)\n        x = layers.Dropout(0.1)(x)\n        x = layers.Add()([residual, x])\n\n    # -----------------------------------------------------------------------\n    # Stage 3: BiGRU (Automatically receives mask from MaskableConv1D)\n    # -----------------------------------------------------------------------\n    x = layers.Bidirectional(\n        layers.GRU(dim // 2, return_sequences=True, dropout=0.1, reset_after=False),\n        name='bigru_1'\n    )(x)\n    x = layers.LayerNormalization(epsilon=1e-6)(x)\n\n    x = layers.Bidirectional(\n        layers.GRU(dim // 4, return_sequences=True, dropout=0.1, reset_after=False),\n        name='bigru_2'\n    )(x)\n    x = layers.LayerNormalization(epsilon=1e-6)(x)\n\n    # -----------------------------------------------------------------------\n    # Pooling & Classification Head\n    # -----------------------------------------------------------------------\n    x = layers.GlobalAveragePooling1D()(x)\n\n    x = layers.Dense(dim, activation='swish', name='head_dense')(x)\n    x = layers.LayerNormalization(epsilon=1e-6)(x)\n    x = layers.Dropout(0.3, name='top_dropout')(x)\n\n    outputs = layers.Dense(num_classes, name='classifier')(x)\n\n    model = models.Model(inputs=inp, outputs=outputs, name='Optimized_Hybrid_Model')\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:24.085619Z","iopub.execute_input":"2026-04-02T04:45:24.08598Z","iopub.status.idle":"2026-04-02T04:45:24.099898Z","shell.execute_reply.started":"2026-04-02T04:45:24.085949Z","shell.execute_reply":"2026-04-02T04:45:24.099163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Build the model using the imported script and global dimensions\nmodel = get_model(max_len=MAX_LEN, channels=CHANNELS, num_classes=NUM_CLASSES)\nmodel_name = model.name\n\nprint(f\" Successfully loaded architecture: {model_name}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:26.046817Z","iopub.execute_input":"2026-04-02T04:45:26.047231Z","iopub.status.idle":"2026-04-02T04:45:26.486541Z","shell.execute_reply.started":"2026-04-02T04:45:26.047197Z","shell.execute_reply":"2026-04-02T04:45:26.485769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import metrics, losses, optimizers\nimport numpy as np\n\nprint(\"=\" * 80)\nprint(\"CONFIGURING MODEL TRAINING PARAMETERS\")\nprint(\"=\" * 80)\n\nsteps_per_epoch = len(train_df_split) // BATCH_SIZE\ntotal_epochs = 100\ntotal_steps = steps_per_epoch * total_epochs\nwarmup_epochs = 3\nwarmup_steps = steps_per_epoch * warmup_epochs\n\nprint(f\"\\nTraining Configuration:\")\nprint(f\"   Steps per epoch:  {steps_per_epoch}\")\nprint(f\"   Total epochs:     {total_epochs}\")\nprint(f\"   Warmup epochs:    {warmup_epochs}\")\nprint(f\"   Total steps:      {total_steps}\")\nprint(f\"   Warmup steps:     {warmup_steps}\")\n\nclass WarmupCosineDecay(tf.keras.optimizers.schedules.LearningRateSchedule):\n    \"\"\"Learning rate schedule with linear warmup and cosine decay.\"\"\"\n    def __init__(self, base_lr, warmup_steps, total_steps, min_lr=1e-6):\n        super().__init__()\n        self.base_lr = tf.constant(base_lr, dtype=tf.float32)\n        self.warmup_steps = tf.constant(warmup_steps, dtype=tf.float32)\n        self.total_steps = tf.constant(total_steps, dtype=tf.float32)\n        self.min_lr = tf.constant(min_lr, dtype=tf.float32)\n\n    def __call__(self, step):\n        step = tf.cast(step, tf.float32)\n        \n        warmup_lr = self.base_lr * (step / self.warmup_steps)\n        \n        progress = (step - self.warmup_steps) / (self.total_steps - self.warmup_steps)\n        progress = tf.clip_by_value(progress, 0.0, 1.0)\n        cosine_lr = self.min_lr + 0.5 * (self.base_lr - self.min_lr) * (\n            1.0 + tf.cos(np.pi * progress)\n        )\n        \n        lr = tf.where(step < self.warmup_steps, warmup_lr, cosine_lr)\n        \n        return lr\n\n    def get_config(self):\n        return {\n            \"base_lr\": float(self.base_lr.numpy()),\n            \"warmup_steps\": int(self.warmup_steps.numpy()),\n            \"total_steps\": int(self.total_steps.numpy()),\n            \"min_lr\": float(self.min_lr.numpy()),\n        }\n\nschedule = WarmupCosineDecay(\n    base_lr=3e-4,\n    warmup_steps=warmup_steps,\n    total_steps=total_steps,\n    min_lr=1e-6\n)\n\nprint(f\"\\nLearning Rate Schedule:\")\ntest_steps = [0, warmup_steps // 2, warmup_steps, total_steps // 2, total_steps]\nfor step in test_steps:\n    lr = schedule(step).numpy()\n    print(f\"   Step {int(step):7d}: LR = {lr:.6f}\")\n\noptimizer = optimizers.Adam(\n    learning_rate=schedule,\n    beta_1=0.9,\n    beta_2=0.999,\n    epsilon=1e-7,\n    clipnorm=1.0,\n    clipvalue=None,\n    decay=0.0,\n    amsgrad=True\n)\n\nprint(f\"\\nOptimizer Configuration:\")\nprint(f\"   Type:           Adam (AMSGrad)\")\nprint(f\"   Learning rate:  Cosine warmup schedule (3e-4 -> 1e-6)\")\nprint(f\"   Gradient clip:  norm=1.0 (prevents exploding gradients)\")\nprint(f\"   Beta 1 (momentum): 0.9\")\nprint(f\"   Beta 2 (RMSprop):  0.999\")\nprint(f\"   AMSGrad:        True (more stable convergence)\")\n\nloss_fn = losses.CategoricalCrossentropy(\n    from_logits=True,\n    label_smoothing=0.0\n)\n\nprint(f\"\\nLoss Function:\")\nprint(f\"   Type:            Categorical Crossentropy\")\nprint(f\"   From logits:     True\")\nprint(f\"   Label smoothing: 0.0 (disabled)\")\n\nmodel.compile(\n    optimizer=optimizer,\n    loss=loss_fn,\n    metrics=[\n        metrics.CategoricalAccuracy(name=\"accuracy\"),\n        metrics.TopKCategoricalAccuracy(k=5, name=\"top_5_accuracy\"),\n        metrics.TopKCategoricalAccuracy(k=10, name=\"top_10_accuracy\"),\n    ]\n)\n\nprint(f\"\\n{'='*80}\")\nprint(f\"MODEL COMPILED SUCCESSFULLY\")\nprint(f\"{'='*80}\")\n\nprint(f\"\\nModel Summary:\")\nprint(f\"   Name:              {model.name}\")\nprint(f\"   Total parameters:  {model.count_params():,}\")\n\ntrainable_params = sum([tf.keras.backend.count_params(w) for w in model.trainable_weights])\nprint(f\"   Trainable params:  {trainable_params:,}\")\n\nnon_trainable_params = model.count_params() - trainable_params\nprint(f\"   Non-trainable:     {non_trainable_params:,}\")\n\nnum_layers = len(model.layers)\nprint(f\"   Number of layers:  {num_layers}\")\n\nprint(f\"\\n{'='*80}\\n\")\n\nmodel.summary(expand_nested=True)\n\nprint(f\"\\n{'='*80}\")\nprint(f\"TRAINING CONFIGURATION SUMMARY\")\nprint(f\"{'='*80}\")\n\nprint(f\"\\nHyperparameters:\")\nprint(f\"   Base Learning Rate:    3e-4\")\nprint(f\"   Min Learning Rate:     1e-6\")\nprint(f\"   Warmup Epochs:         3\")\nprint(f\"   Total Epochs:          100\")\nprint(f\"   Batch Size:            {BATCH_SIZE}\")\nprint(f\"   Optimizer:             Adam with AMSGrad\")\nprint(f\"   Loss:                  Categorical Crossentropy (from_logits)\")\nprint(f\"   Gradient Clip Norm:    1.0\")\n\nprint(f\"\\nExpected Training Behavior:\")\nprint(f\"   Epoch 0-2:    Loss decreases rapidly, accuracy ~0.3-0.4\")\nprint(f\"   Epoch 3-20:   Steady improvement, LR decay starts\")\nprint(f\"   Epoch 20-50:  Diminishing returns, validation plateau\")\nprint(f\"   Epoch 50+:    Fine-tuning, avoid overfitting\")\n\nprint(f\"\\n{'='*80}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:31.52757Z","iopub.execute_input":"2026-04-02T04:45:31.528362Z","iopub.status.idle":"2026-04-02T04:45:33.43595Z","shell.execute_reply.started":"2026-04-02T04:45:31.528327Z","shell.execute_reply":"2026-04-02T04:45:33.435162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras.utils import plot_model\n\nos.makedirs(BASE_DIR, exist_ok=True)\n\n# 1. Save Architecture Diagram\narchitecture_path = os.path.join(BASE_DIR, f\"{model_name}_architecture.png\")\ntry:\n    plot_model(model, to_file=architecture_path, show_shapes=True, show_layer_names=True, expand_nested=True, dpi=300)\n    print(f\"Architecture diagram saved: {architecture_path}\")\nexcept Exception as e:\n    print(\"Note: Install 'pydot' and 'graphviz' to generate architecture plots.\")\n\n# 2. Log Model Parameters\nparams = model.count_params()\nparams_path = os.path.join(BASE_DIR, f\"{model_name}_params.txt\")\nwith open(params_path, \"w\") as f:\n    f.write(f\"Total Parameters: {params:,}\\n\")\n    f.write(f\"Trainable Parameters: {sum([tf.keras.backend.count_params(w) for w in model.trainable_weights]):,}\\n\")\nprint(f\"Total Parameters: {params:,} (Saved to {params_path})\")\n\n# 3. Define Custom Callbacks\nclass TrainingPlot(callbacks.Callback):\n    def __init__(self, model_name):\n        super().__init__()\n        self.model_name = model_name\n        self.history_dict = {'accuracy': [], 'val_accuracy': [], 'loss': [], 'val_loss': []}\n\n    def on_epoch_end(self, epoch, logs=None):\n        for key in self.history_dict.keys():\n            if key in logs:\n                self.history_dict[key].append(logs[key])\n\n    def on_train_end(self, logs=None):\n        # Accuracy Plot\n        plt.figure(figsize=(10, 5))\n        plt.plot(self.history_dict.get('accuracy', []))\n        plt.plot(self.history_dict.get('val_accuracy', []))\n        plt.title(f\"{self.model_name} - Accuracy Curve\", fontweight='bold')\n        plt.xlabel(\"Epoch\")\n        plt.ylabel(\"Accuracy\")\n        plt.legend([\"Train\", \"Validation\"])\n        plt.grid(True, linestyle='--', alpha=0.7)\n        plt.savefig(os.path.join(BASE_DIR, f\"{self.model_name}_accuracy_curve.png\"), dpi=300, bbox_inches='tight')\n        plt.close()\n\n        # Loss Plot\n        plt.figure(figsize=(10, 5))\n        plt.plot(self.history_dict.get('loss', []))\n        plt.plot(self.history_dict.get('val_loss', []))\n        plt.title(f\"{self.model_name} - Loss Curve\", fontweight='bold')\n        plt.xlabel(\"Epoch\")\n        plt.ylabel(\"Loss\")\n        plt.legend([\"Train\", \"Validation\"])\n        plt.grid(True, linestyle='--', alpha=0.7)\n        plt.savefig(os.path.join(BASE_DIR, f\"{self.model_name}_loss_curve.png\"), dpi=300, bbox_inches='tight')\n        plt.close()\n\nclass BestMetricLogger(callbacks.Callback):\n    def __init__(self, model_name):\n        super().__init__()\n        self.model_name = model_name\n        self.best_val_acc = 0.0\n\n    def on_epoch_end(self, epoch, logs=None):\n        current_val_acc = logs.get('val_accuracy', 0.0)\n        if current_val_acc > self.best_val_acc:\n            self.best_val_acc = current_val_acc\n\n    def on_train_end(self, logs=None):\n        with open(os.path.join(BASE_DIR, f\"{self.model_name}_best_score.txt\"), \"w\") as f:\n            f.write(f\"Best Validation Accuracy: {self.best_val_acc:.4f}\\n\")\n\nclass SaveLastEpoch(callbacks.Callback):\n    \"\"\"Saves model after every epoch — allows resume if session dies.\"\"\"\n    def on_epoch_end(self, epoch, logs=None):\n        self.model.save(os.path.join(BASE_DIR, f\"{model_name}_last.keras\"))\n\n# 4. Resume from checkpoint if exists\nLAST_PATH = os.path.join(BASE_DIR, f\"{model_name}_last.keras\")\ninitial_epoch = 0\n\nif os.path.exists(LAST_PATH):\n    model.load_weights(LAST_PATH)\n    log_path = os.path.join(BASE_DIR, f\"{model_name}_log.csv\")\n    if os.path.exists(log_path):\n        import pandas as pd\n        log_df = pd.read_csv(log_path)\n        initial_epoch = len(log_df)\n    print(f\"Resumed from epoch {initial_epoch}\")\nelse:\n    print(\"Starting fresh training.\")\n\n# 5. Build Final Callbacks List\nmodel_callbacks = [\n    callbacks.ModelCheckpoint(\n        filepath=os.path.join(BASE_DIR, f\"{model_name}_best.keras\"),  # CHANGED: .h5 → .keras\n        monitor='val_accuracy',\n        save_best_only=True,\n        mode='max',\n        verbose=1\n    ),\n    callbacks.EarlyStopping(\n        monitor='val_accuracy',\n        patience=20,\n        restore_best_weights=True,\n        verbose=1\n    ),\n    # REMOVED: ReduceLROnPlateau — conflicts with WarmupCosineDecay schedule\n    callbacks.CSVLogger(\n        filename=os.path.join(BASE_DIR, f\"{model_name}_log.csv\"),\n        append=True         # CHANGED: False → True عشان يضيف على السجل القديم مش يمسحه\n    ),\n    TrainingPlot(model_name),\n    BestMetricLogger(model_name),\n    callbacks.TensorBoard(log_dir=os.path.join(BASE_DIR, model_name), histogram_freq=0),\n    SaveLastEpoch(),        # ADDED: بيحفظ بعد كل epoch للـ resume\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:42.584886Z","iopub.execute_input":"2026-04-02T04:45:42.585581Z","iopub.status.idle":"2026-04-02T04:45:50.403969Z","shell.execute_reply.started":"2026-04-02T04:45:42.585546Z","shell.execute_reply":"2026-04-02T04:45:50.402946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"steps_per_epoch  = len(train_df_split) // BATCH_SIZE\nvalidation_steps = len(val_df_split)   // BATCH_SIZE\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:50.405592Z","iopub.execute_input":"2026-04-02T04:45:50.406041Z","iopub.status.idle":"2026-04-02T04:45:50.410292Z","shell.execute_reply.started":"2026-04-02T04:45:50.405964Z","shell.execute_reply":"2026-04-02T04:45:50.40935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Define the maximum number of epochs\n# (EarlyStopping will likely stop it much earlier, usually around 30-50 epochs)\nTRAINING_EPOCHS = 100\n\nprint(f\"Maximum Epochs set to: {TRAINING_EPOCHS}\")\nprint(f\"Batch Size (handled by tf.data): {BATCH_SIZE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:50.411259Z","iopub.execute_input":"2026-04-02T04:45:50.411568Z","iopub.status.idle":"2026-04-02T04:45:50.427281Z","shell.execute_reply.started":"2026-04-02T04:45:50.41153Z","shell.execute_reply":"2026-04-02T04:45:50.426462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rebuild datasets with .repeat() applied\ntrain_dataset = get_parquet_dataset(\n    train_df_split,\n    data_dir=DATA_DIR,\n    batch_size=BATCH_SIZE,\n    max_len=MAX_LEN,\n    augment=True,\n    shuffle=True\n)\n\nval_dataset = get_parquet_dataset(\n    val_df_split,\n    data_dir=DATA_DIR,\n    batch_size=BATCH_SIZE,\n    max_len=MAX_LEN,\n    augment=False,\n    shuffle=False\n)\n\nsteps_per_epoch  = len(train_df_split) // BATCH_SIZE\nvalidation_steps = len(val_df_split)   // BATCH_SIZE\n\nprint(f\"steps_per_epoch  : {steps_per_epoch}\")\nprint(f\"validation_steps : {validation_steps}\")\nprint(\"Datasets rebuilt successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:53.944531Z","iopub.execute_input":"2026-04-02T04:45:53.944873Z","iopub.status.idle":"2026-04-02T04:45:54.531689Z","shell.execute_reply.started":"2026-04-02T04:45:53.944842Z","shell.execute_reply":"2026-04-02T04:45:54.530898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Training for: {model_name}...\")\n\nsteps_per_epoch  = len(train_df_split) // BATCH_SIZE\nvalidation_steps = len(val_df_split)   // BATCH_SIZE\n\nhistory = model.fit(\n    train_dataset,\n    validation_data=val_dataset,\n    epochs=TRAINING_EPOCHS,\n    initial_epoch=initial_epoch,          \n    steps_per_epoch=steps_per_epoch,\n    validation_steps=validation_steps,\n    callbacks=model_callbacks,\n    verbose=1\n)\n\nprint(f\"\\n Training Phase Completed for {model_name}!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:45:57.407313Z","iopub.execute_input":"2026-04-02T04:45:57.408117Z","iopub.status.idle":"2026-04-02T04:48:33.307295Z","shell.execute_reply.started":"2026-04-02T04:45:57.408083Z","shell.execute_reply":"2026-04-02T04:48:33.305317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1. Training History\n| Goal | Outputs |\n| :--- | :--- |\n| Trace the training process and detect overfitting/underfitting. | Accuracy and Loss curves over epochs. |","metadata":{}},{"cell_type":"code","source":"print(f\"--- Visualizing Training History for {model_name} ---\")\n\n# Load training history from the saved CSV log\nlog_path = os.path.join(BASE_DIR, f\"{model_name}_log.csv\")\n\nif os.path.exists(log_path):\n    history_df = pd.read_csv(log_path)\n    \n    fig, axes = plt.subplots(1, 2, figsize=(16, 5))\n    \n    # Accuracy Plot\n    axes[0].plot(history_df['epoch'], history_df['accuracy'], label='Train Accuracy', color='#4285F4', linewidth=2)\n    axes[0].plot(history_df['epoch'], history_df['val_accuracy'], label='Validation Accuracy', color='#34A853', linewidth=2)\n    axes[0].set_title('Model Accuracy over Epochs', fontweight='bold')\n    axes[0].set_xlabel('Epoch')\n    axes[0].set_ylabel('Accuracy')\n    axes[0].legend()\n    axes[0].grid(True, linestyle='--', alpha=0.6)\n    \n    # Loss Plot\n    axes[1].plot(history_df['epoch'], history_df['loss'], label='Train Loss', color='#EA4335', linewidth=2)\n    axes[1].plot(history_df['epoch'], history_df['val_loss'], label='Validation Loss', color='#FBBC05', linewidth=2)\n    axes[1].set_title('Model Loss over Epochs', fontweight='bold')\n    axes[1].set_xlabel('Epoch')\n    axes[1].set_ylabel('Loss')\n    axes[1].legend()\n    axes[1].grid(True, linestyle='--', alpha=0.6)\n    \n    plt.tight_layout()\n    plt.savefig(os.path.join(BASE_DIR, f\"{model_name}_Training_History.png\"), dpi=300)\n    plt.show()\nelse:\n    print(f\"Log file not found at {log_path}. Ensure the model has been trained.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.126361Z","iopub.status.idle":"2026-04-02T04:40:55.126704Z","shell.execute_reply.started":"2026-04-02T04:40:55.126565Z","shell.execute_reply":"2026-04-02T04:40:55.126584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Comprehensive Evaluation\n| Goal | Outputs |\n| :--- | :--- |\n| Comprehensive assessment of the model on entirely unseen data. | Overall Loss, Accuracy, and Top-5 Accuracy scores. |","metadata":{}},{"cell_type":"code","source":"print(f\"--- Executing Comprehensive Evaluation for {model_name} ---\")\n\n# Evaluate directly on the optimized test_dataset\neval_metrics = model.evaluate(test_dataset, verbose=1)\n\nprint(\"\\n=========================================================\")\nprint(\"             OVERALL TEST SET METRICS\")\nprint(\"=========================================================\")\nprint(f\"Test Loss            : {eval_metrics[0]:.4f}\")\nprint(f\"Test Accuracy        : {eval_metrics[1]:.4f}  ({eval_metrics[1]*100:.2f}%)\")\nprint(f\"Test Top-5 Accuracy  : {eval_metrics[2]:.4f}  ({eval_metrics[2]*100:.2f}%)\")\nprint(\"=========================================================\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.127806Z","iopub.status.idle":"2026-04-02T04:40:55.128129Z","shell.execute_reply.started":"2026-04-02T04:40:55.127953Z","shell.execute_reply":"2026-04-02T04:40:55.127969Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3. Classification Report\n| Goal | Outputs |\n| :--- | :--- |\n| Calculate foundational class-wise metrics. | CSV and TXT reports containing Precision, Recall, and F1-Score for all classes. |","metadata":{}},{"cell_type":"code","source":"print(\"--- Generating Predictions and Classification Report ---\")\n\ny_true = []\ny_pred_probs = []\n\n# Extract true labels and compute predicted probabilities\nfor x_batch, y_batch in test_dataset:\n    preds = model.predict(x_batch, verbose=0)\n    y_pred_probs.extend(preds)\n    y_true.extend(np.argmax(y_batch.numpy(), axis=1))\n\ny_true = np.array(y_true)\ny_pred_probs = np.array(y_pred_probs)\ny_pred = np.argmax(y_pred_probs, axis=1)\n\n# Generate Class Names mapping\ntarget_names = [label_to_sign[i] for i in range(NUM_CLASSES)] if 'label_to_sign' in globals() else [str(i) for i in range(NUM_CLASSES)]\n\n# Compute Classification Report\nreport_dict = classification_report(y_true, y_pred, target_names=target_names, output_dict=True, zero_division=0)\nreport_df = pd.DataFrame(report_dict).transpose()\n\n# Export to CSV\ncsv_path = os.path.join(BASE_DIR, f\"{model_name}_Classification_Report.csv\")\nreport_df.to_csv(csv_path)\n\n# Print Summary\nprint(f\"Full report saved to: {csv_path}\")\nprint(\"\\nGlobal Averages:\")\nprint(f\"-> Macro Avg F1-Score   : {report_dict['macro avg']['f1-score']:.4f}\")\nprint(f\"-> Weighted Avg F1-Score: {report_dict['weighted avg']['f1-score']:.4f}\")\n\n# Display top 5 rows of the dataframe\ndisplay(report_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.12942Z","iopub.status.idle":"2026-04-02T04:40:55.12979Z","shell.execute_reply.started":"2026-04-02T04:40:55.129565Z","shell.execute_reply":"2026-04-02T04:40:55.129587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 4. Confusion Matrix\n| Goal | Outputs |\n| :--- | :--- |\n| Understand specific model misclassifications and overlaps. | Absolute and Normalized Confusion Matrix high-resolution images. |","metadata":{}},{"cell_type":"code","source":"print(\"--- Plotting Absolute and Normalized Confusion Matrices ---\")\n\n# 1. Absolute Confusion Matrix\ncm_absolute = confusion_matrix(y_true, y_pred)\nplt.figure(figsize=(24, 20))\nsns.heatmap(cm_absolute, cmap=\"Blues\", cbar=True, xticklabels=False, yticklabels=False)\nplt.title(f\"{model_name} - Absolute Confusion Matrix\", fontsize=22, pad=20)\nplt.xlabel(\"Predicted Sign Class\", fontsize=16)\nplt.ylabel(\"True Sign Class\", fontsize=16)\nabs_path = os.path.join(\"../Evaluation_Plots\", f\"{model_name}_CM_Absolute.png\")\nplt.savefig(abs_path, dpi=300, bbox_inches='tight')\nplt.close()\n\n# 2. Normalized Confusion Matrix (Percentages)\ncm_normalized = confusion_matrix(y_true, y_pred, normalize='true')\nplt.figure(figsize=(24, 20))\nsns.heatmap(cm_normalized, cmap=\"rocket_r\", cbar=True, xticklabels=False, yticklabels=False)\nplt.title(f\"{model_name} - Normalized Confusion Matrix (Recall per Class)\", fontsize=22, pad=20)\nplt.xlabel(\"Predicted Sign Class\", fontsize=16)\nplt.ylabel(\"True Sign Class\", fontsize=16)\nnorm_path = os.path.join(BASE_DIR, f\"{model_name}_CM_Normalized.png\")\nplt.savefig(norm_path, dpi=300, bbox_inches='tight')\nplt.close()\n\nprint(f\" Absolute CM saved to: {abs_path}\")\nprint(f\" Normalized CM saved to: {norm_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.13109Z","iopub.status.idle":"2026-04-02T04:40:55.131369Z","shell.execute_reply.started":"2026-04-02T04:40:55.131226Z","shell.execute_reply":"2026-04-02T04:40:55.131242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 5. Class Performance\n| Goal | Outputs |\n| :--- | :--- |\n| Analyze the individual performance of each sign class. | Horizontal bar charts highlighting the Top 20 Best and Top 20 Worst performing classes. |","metadata":{}},{"cell_type":"code","source":"print(\"--- Analyzing Best and Worst Performing Classes ---\")\n\n# Extract only the 250 classes (remove 'accuracy', 'macro avg', 'weighted avg')\nclass_metrics = report_df.iloc[:-3].copy()\nclass_metrics = class_metrics.sort_values(by='f1-score', ascending=False)\n\nbest_20 = class_metrics.head(20)\nworst_20 = class_metrics.tail(20).sort_values(by='f1-score', ascending=True)\n\nfig, axes = plt.subplots(1, 2, figsize=(20, 10))\n\n# Best 20 Plot\naxes[0].barh(best_20.index[::-1], best_20['f1-score'][::-1], color='#34A853')\naxes[0].set_title('Top 20 Performing Signs (Highest F1-Score)', fontsize=16, fontweight='bold')\naxes[0].set_xlabel('F1-Score', fontsize=12)\naxes[0].set_xlim(0, 1.05)\naxes[0].grid(axis='x', linestyle='--', alpha=0.5)\n\n# Worst 20 Plot\naxes[1].barh(worst_20.index, worst_20['f1-score'], color='#EA4335')\naxes[1].set_title('Top 20 Most Challenging Signs (Lowest F1-Score)', fontsize=16, fontweight='bold')\naxes[1].set_xlabel('F1-Score', fontsize=12)\naxes[1].set_xlim(0, 1.05)\naxes[1].grid(axis='x', linestyle='--', alpha=0.5)\n\nplt.tight_layout()\nperf_path = os.path.join(BASE_DIR, f\"{model_name}_Class_Performance.png\")\nplt.savefig(perf_path, dpi=300)\nplt.show()\nprint(f\" Class performance charts saved to: {perf_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.13248Z","iopub.status.idle":"2026-04-02T04:40:55.13285Z","shell.execute_reply.started":"2026-04-02T04:40:55.132644Z","shell.execute_reply":"2026-04-02T04:40:55.132668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6. Metrics Distribution\n| Goal | Outputs |\n| :--- | :--- |\n| Visualize the statistical distribution of evaluation metrics across all classes. | Histograms detailing the density of Precision, Recall, and F1-Scores. |","metadata":{}},{"cell_type":"code","source":"print(\"--- Plotting Metrics Distribution Across All Classes ---\")\n\nclass_metrics = report_df.iloc[:-3]\n\nplt.figure(figsize=(12, 6))\nsns.kdeplot(class_metrics['precision'], fill=True, label='Precision', color='#4285F4', alpha=0.4)\nsns.kdeplot(class_metrics['recall'], fill=True, label='Recall', color='#FBBC05', alpha=0.4)\nsns.kdeplot(class_metrics['f1-score'], fill=True, label='F1-Score', color='#34A853', alpha=0.4)\n\nplt.title(\"Distribution of Evaluation Metrics Across 250 Classes\", fontsize=16, fontweight='bold', pad=15)\nplt.xlabel(\"Metric Score\", fontsize=14)\nplt.ylabel(\"Density\", fontsize=14)\nplt.legend(fontsize=12)\nplt.xlim(0, 1)\nplt.grid(True, linestyle='--', alpha=0.5)\n\ndist_path = os.path.join(BASE_DIR, f\"{model_name}_Metrics_Distribution.png\")\nplt.savefig(dist_path, dpi=300)\nplt.show()\nprint(f\"Distribution plot saved to: {dist_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.13496Z","iopub.status.idle":"2026-04-02T04:40:55.135406Z","shell.execute_reply.started":"2026-04-02T04:40:55.135262Z","shell.execute_reply":"2026-04-02T04:40:55.13528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 7. ROC Curves\n| Goal | Outputs |\n| :--- | :--- |\n| Measure the discrimination ability of the classifier. | Receiver Operating Characteristic (ROC) curves and Area Under Curve (AUC) values for the macro-average and top challenging classes. |","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import label_binarize\nfrom sklearn.metrics import roc_curve, auc\n\nprint(\"--- Computing ROC Curves and AUC ---\")\n\n# Binarize labels for multi-class ROC calculation\ny_true_bin = label_binarize(y_true, classes=range(NUM_CLASSES))\n\n# Compute micro-average ROC curve and ROC area\nfpr_micro, tpr_micro, _ = roc_curve(y_true_bin.ravel(), y_pred_probs.ravel())\nroc_auc_micro = auc(fpr_micro, tpr_micro)\n\nplt.figure(figsize=(10, 8))\nplt.plot(fpr_micro, tpr_micro, label=f'Micro-average ROC (AUC = {roc_auc_micro:.3f})', \n         color='deeppink', linestyle=':', linewidth=4)\n\n# Plot standard diagonal line\nplt.plot([0, 1], [0, 1], 'k--', linewidth=2)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate', fontsize=14)\nplt.ylabel('True Positive Rate', fontsize=14)\nplt.title(f'{model_name} - ROC Curve (Micro Average)', fontsize=16, fontweight='bold')\nplt.legend(loc=\"lower right\", fontsize=12)\nplt.grid(True, linestyle='--', alpha=0.5)\n\nroc_path = os.path.join(BASE_DIR, f\"{model_name}_ROC_Curve.png\")\nplt.savefig(roc_path, dpi=300)\nplt.show()\nprint(f\" ROC Curve saved to: {roc_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.136778Z","iopub.status.idle":"2026-04-02T04:40:55.13716Z","shell.execute_reply.started":"2026-04-02T04:40:55.136939Z","shell.execute_reply":"2026-04-02T04:40:55.136961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 8. PR Curves\n| Goal | Outputs |\n| :--- | :--- |\n| Evaluate model capability focusing on positive class prediction, especially useful for any inherent data imbalances. | Precision-Recall (PR) curves and Average Precision (AP) scores. |","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve, average_precision_score\n\nprint(\"--- Computing Precision-Recall Curves ---\")\n\nprecision_micro, recall_micro, _ = precision_recall_curve(y_true_bin.ravel(), y_pred_probs.ravel())\nap_micro = average_precision_score(y_true_bin, y_pred_probs, average=\"micro\")\n\nplt.figure(figsize=(10, 8))\nplt.plot(recall_micro, precision_micro, label=f'Micro-average PR (AP = {ap_micro:.3f})', \n         color='navy', linestyle='-', linewidth=3)\n\nplt.xlabel('Recall', fontsize=14)\nplt.ylabel('Precision', fontsize=14)\nplt.title(f'{model_name} - Precision-Recall Curve (Micro Average)', fontsize=16, fontweight='bold')\nplt.legend(loc=\"lower left\", fontsize=12)\nplt.grid(True, linestyle='--', alpha=0.5)\n\npr_path = os.path.join(BASE_DIR, f\"{model_name}_PR_Curve.png\")\nplt.savefig(pr_path, dpi=300)\nplt.show()\nprint(f\" PR Curve saved to: {pr_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.139275Z","iopub.status.idle":"2026-04-02T04:40:55.139634Z","shell.execute_reply.started":"2026-04-02T04:40:55.139495Z","shell.execute_reply":"2026-04-02T04:40:55.139514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 9. Error Analysis\n| Goal | Outputs |\n| :--- | :--- |\n| Conduct a deep dive into common misclassifications to understand the root causes of confusion. | Identification and visualization of the most frequently confused sign pairs. |","metadata":{}},{"cell_type":"code","source":"import collections\n\nprint(\"--- Conducting Deep Error Analysis (Most Confused Pairs) ---\")\n\n# Identify all misclassified instances\nmisclassified_indices = np.where(y_true != y_pred)[0]\n\n# Create pairs of (True Label, Predicted Label)\nerror_pairs = [(target_names[y_true[i]], target_names[y_pred[i]]) for i in misclassified_indices]\n\n# Count the frequency of each specific error pair\nerror_counts = collections.Counter(error_pairs)\ntop_errors = error_counts.most_common(10)\n\n# Format for plotting\nerror_labels = [f\"True: {pair[0][0]}\\nPred: {pair[0][1]}\" for pair in top_errors]\nerror_values = [pair[1] for pair in top_errors]\n\nplt.figure(figsize=(14, 7))\nsns.barplot(x=error_values, y=error_labels, palette=\"Reds_r\")\nplt.title(f\"{model_name} - Top 10 Most Frequently Confused Sign Pairs\", fontsize=16, fontweight='bold', pad=15)\nplt.xlabel(\"Number of Misclassifications\", fontsize=14)\nplt.ylabel(\"Confusion Pair\", fontsize=14)\n\n# Add value labels\nfor index, value in enumerate(error_values):\n    plt.text(value + 0.5, index, str(value), va='center', fontsize=12, fontweight='bold')\n\nplt.tight_layout()\nerror_path = os.path.join(BASE_DIR, f\"{model_name}_Top_Confusions.png\")\nplt.savefig(error_path, dpi=300)\nplt.show()\nprint(f\" Error analysis chart saved to: {error_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.141145Z","iopub.status.idle":"2026-04-02T04:40:55.141564Z","shell.execute_reply.started":"2026-04-02T04:40:55.14136Z","shell.execute_reply":"2026-04-02T04:40:55.141384Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  Classification Report\n","metadata":{}},{"cell_type":"code","source":"print(\"--- Extracting Predictions from Test Dataset ---\")\n\ny_true = []\ny_pred_probs = []\n\n# Iterate through the test dataset to gather true labels and model predictions\nfor x_batch, y_batch in test_dataset:\n    preds = model.predict(x_batch, verbose=0)\n    y_pred_probs.extend(preds)\n    y_true.extend(np.argmax(y_batch.numpy(), axis=1))\n\ny_true = np.array(y_true)\ny_pred_probs = np.array(y_pred_probs)\ny_pred = np.argmax(y_pred_probs, axis=1)\n\nprint(\"Predictions extracted successfully.\")\n\n# Generate the classification report\nprint(\"\\n--- Generating Classification Report ---\")\n# If label_to_sign mapping exists, use it; otherwise use numeric indices\ntarget_names = [label_to_sign[i] for i in range(NUM_CLASSES)] if 'label_to_sign' in globals() else [str(i) for i in range(NUM_CLASSES)]\n\nreport_dict = classification_report(y_true, y_pred, target_names=target_names, output_dict=True, zero_division=0)\nreport_df = pd.DataFrame(report_dict).transpose()\n\n# Save report to CSV\nreport_csv_path = os.path.join(BASE_DIR, f\"{model_name}_Classification_Report.csv\")\nreport_df.to_csv(report_csv_path)\n\nprint(f\"Classification report saved to: {report_csv_path}\")\n\n# Display macro and weighted averages\nprint(\"\\nGlobal Metrics:\")\nprint(f\"Macro Avg F1-Score   : {report_dict['macro avg']['f1-score']:.4f}\")\nprint(f\"Weighted Avg F1-Score: {report_dict['weighted avg']['f1-score']:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.143099Z","iopub.status.idle":"2026-04-02T04:40:55.14344Z","shell.execute_reply.started":"2026-04-02T04:40:55.143299Z","shell.execute_reply":"2026-04-02T04:40:55.143321Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Error Analysis (Lowest Performing Classes)","metadata":{}},{"cell_type":"code","source":"\n\n# Extract F1-scores for individual classes (excluding macro/weighted avgs)\nclass_metrics = report_df.iloc[:-3]\nclass_metrics = class_metrics.sort_values(by='f1-score', ascending=True)\n\n# Select the bottom 20 performing classes\nbottom_20 = class_metrics.head(20)\n\nplt.figure(figsize=(12, 8))\nplt.barh(bottom_20.index, bottom_20['f1-score'], color='#E53935')\n\nplt.title(f\"{model_name} - Top 20 Most Challenging Signs (Lowest F1-Score)\", fontsize=16, fontweight='bold')\nplt.xlabel(\"F1-Score\", fontsize=12)\nplt.ylabel(\"Sign Class\", fontsize=12)\nplt.xlim(0, 1.0)\nplt.grid(axis='x', linestyle='--', alpha=0.7)\n\n# Add exact numbers on the bars\nfor index, value in enumerate(bottom_20['f1-score']):\n    plt.text(value + 0.01, index, f\"{value:.2f}\", va='center', fontsize=10)\n\nplt.tight_layout()\n\n# Save the plot\nworst_classes_path = os.path.join(BASE_DIR, f\"{model_name}_Worst_Classes.png\")\nplt.savefig(worst_classes_path, dpi=300)\nplt.close()\n\nprint(f\"Error analysis plot saved to: {worst_classes_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.144597Z","iopub.status.idle":"2026-04-02T04:40:55.144954Z","shell.execute_reply.started":"2026-04-02T04:40:55.144742Z","shell.execute_reply":"2026-04-02T04:40:55.144765Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import subprocess\nimport sys\n\n# Run without --quiet to see the actual error\nresult = subprocess.run([\n    \"pip\", \"install\",\n    \"protobuf==3.19.6\",\n    \"onnx==1.13.0\",\n    \"tf2onnx==1.14.0\",\n    \"--force-reinstall\"\n], capture_output=True, text=True)\n\nprint(\"STDOUT:\")\nprint(result.stdout)\nprint(\"STDERR:\")\nprint(result.stderr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.14578Z","iopub.status.idle":"2026-04-02T04:40:55.146106Z","shell.execute_reply.started":"2026-04-02T04:40:55.145924Z","shell.execute_reply":"2026-04-02T04:40:55.145941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport tensorflow as tf\n\ndef export_universal_model(model, model_name, saved_dir=BASE_DIR, max_len=384, channels=708):\n    print(f\"\\n=========================================================\")\n    print(f\"   STARTING UNIVERSAL EXPORT FOR: {model_name}\")\n    print(f\"=========================================================\")\n\n    os.makedirs(saved_dir, exist_ok=True)\n\n    # 1. Export Standard Keras Model\n    keras_path = os.path.join(saved_dir, f\"{model_name}_final.keras\")\n    try:\n        model.save(keras_path)\n        print(f\"[1/2]  Standard Keras model saved: {keras_path}\")\n    except Exception as e:\n        print(f\"[1/2]  Failed to save Keras model: {e}\")\n\n    # 2. Export TFLite Model\n    tflite_path = os.path.join(saved_dir, f\"{model_name}.tflite\")\n    try:\n        converter = tf.lite.TFLiteConverter.from_keras_model(model)\n        converter.target_spec.supported_ops = [\n            tf.lite.OpsSet.TFLITE_BUILTINS,\n            tf.lite.OpsSet.SELECT_TF_OPS\n        ]\n        converter._experimental_lower_tensor_list_ops = False\n        converter.optimizations = [tf.lite.Optimize.DEFAULT]\n        tflite_model = converter.convert()\n        with open(tflite_path, 'wb') as f:\n            f.write(tflite_model)\n        print(f\"[2/2]  TFLite model saved (Mobile Ready): {tflite_path}\")\n    except Exception as e:\n        print(f\"[2/2]  Failed to save TFLite model: {e}\")\n\n    # 3. ONNX Export — requires kernel restart, run separately after training\n    print(f\"[SKIP] ONNX export skipped — run export_onnx_after_restart.py after kernel restart\")\n    onnx_script_path = os.path.join(saved_dir, \"export_onnx_after_restart.py\")\n    with open(onnx_script_path, \"w\") as f:\n        f.write(f\"\"\"import tensorflow as tf\nimport tf2onnx\n\nmodel = tf.keras.models.load_model(r\"{keras_path}\")\ninput_signature = [tf.TensorSpec([None, {max_len}, {channels}], tf.float32, name='input_features')]\ntf2onnx.convert.from_keras(model, input_signature=input_signature, opset=13, output_path=r\"{os.path.join(saved_dir, f'{model_name}.onnx')}\")\nprint(\"ONNX export done.\")\n\"\"\")\n    print(f\"[INFO] ONNX script saved to: {onnx_script_path}\")\n    print(f\"       Run it after kernel restart with: python {onnx_script_path}\")\n\n    print(f\"=========================================================\")\n    print(f\"   EXPORT PIPELINE COMPLETED FOR: {model_name}\")\n    print(f\"=========================================================\\n\")\n\nexport_universal_model(model, model_name=model.name, max_len=MAX_LEN, channels=CHANNELS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-02T04:40:55.148008Z","iopub.status.idle":"2026-04-02T04:40:55.148433Z","shell.execute_reply.started":"2026-04-02T04:40:55.148284Z","shell.execute_reply":"2026-04-02T04:40:55.148302Z"}},"outputs":[],"execution_count":null}]}