{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15938270,"datasetId":10221342,"databundleVersionId":16896209},{"sourceType":"modelInstanceVersion","sourceId":870517,"databundleVersionId":17242785,"modelInstanceId":662054,"modelId":673883,"isSourceIdPinned":false},{"sourceType":"modelInstanceVersion","sourceId":846565,"databundleVersionId":16895398,"modelInstanceId":643793,"modelId":655752,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"21d4ac3d","cell_type":"markdown","source":"###  Environment Setup and Library Imports\n","metadata":{}},{"id":"cbac16ca-e472-4f5e-8178-9fef832d0684","cell_type":"code","source":"import tensorflow as tf\n\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n        \n        logical_gpus = tf.config.list_logical_devices('GPU')\n        print(f\"{len(gpus)} Physical GPUs, {len(logical_gpus)} Logical GPUs\")\n    except RuntimeError as e:\n        print(e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:05.684117Z","iopub.execute_input":"2026-04-26T03:40:05.684844Z","iopub.status.idle":"2026-04-26T03:40:51.903093Z","shell.execute_reply.started":"2026-04-26T03:40:05.684808Z","shell.execute_reply":"2026-04-26T03:40:51.902203Z"}},"outputs":[],"execution_count":null},{"id":"16122586-47d5-40bf-9c6e-c78119bbe5d8","cell_type":"code","source":"# =========================================================\n# WESSAL PROJECT - MASTER DIRECTORY BOOTSTRAP (KAGGLE)\n# Run this in FIRST CELL only\n# =========================================================\n\nimport os\nimport shutil\n\n# Root folder\nBASE_DIR = \"/kaggle/working/Wessal_Project\"\n\n# Remove old project completely (optional clean rebuild)\nif os.path.exists(BASE_DIR):\n    shutil.rmtree(BASE_DIR)\n\n# Main folders\nfolders = [\n    \"Evaluation_Plots\",\n    \"Saved_Models\",\n    \"Training_Histories\",\n    \"architecture_model\",\n    \"logs\",\n    \"notebooks\",\n    \"results\"\n]\n\n# Create root\nos.makedirs(BASE_DIR, exist_ok=True)\n\n# Create folders\nfor folder in folders:\n    os.makedirs(os.path.join(BASE_DIR, folder), exist_ok=True)\n\n# Create empty files\nopen(os.path.join(BASE_DIR, \"README.md\"), \"w\").close()\nopen(os.path.join(BASE_DIR, \"requirements.txt\"), \"w\").close()\n\n# =========================================================\n# Global Paths (Use everywhere later)\n# =========================================================\n\nEVAL_DIR         = os.path.join(BASE_DIR, \"Evaluation_Plots\")\nMODELS_DIR       = os.path.join(BASE_DIR, \"Saved_Models\")\nHISTORY_DIR      = os.path.join(BASE_DIR, \"Training_Histories\")\nARCH_DIR         = os.path.join(BASE_DIR, \"architecture_model\")\nLOGS_DIR         = os.path.join(BASE_DIR, \"logs\")\nNOTEBOOKS_DIR    = os.path.join(BASE_DIR, \"notebooks\")\nRESULTS_DIR      = os.path.join(BASE_DIR, \"results\")\n\n# =========================================================\n# Display Tree\n# =========================================================\n\nprint(\"Project Structure Ready:\\n\")\nfor root, dirs, files in os.walk(BASE_DIR):\n    level = root.replace(BASE_DIR, \"\").count(os.sep)\n    indent = \" \" * 4 * level\n    print(f\"{indent}{os.path.basename(root)}/\")\n    subindent = \" \" * 4 * (level + 1)\n    for f in files:\n        print(f\"{subindent}{f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:51.904512Z","iopub.execute_input":"2026-04-26T03:40:51.904951Z","iopub.status.idle":"2026-04-26T03:40:51.914702Z","shell.execute_reply.started":"2026-04-26T03:40:51.904924Z","shell.execute_reply":"2026-04-26T03:40:51.913969Z"}},"outputs":[],"execution_count":null},{"id":"e0b93acc","cell_type":"code","source":"# Standard System Libraries\nimport os\nimport sys\nimport gc\nimport time\nimport math\nimport random\nimport pickle\nimport glob\nimport datetime\n\n# Data Manipulation & Visualization\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\nfrom tqdm.autonotebook import tqdm\n\n# Machine Learning & Evaluation Metrics\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (accuracy_score, precision_score, \n                             recall_score, f1_score, \n                             classification_report, confusion_matrix)\n\n# Deep Learning (TensorFlow & Keras)\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model, load_model\nfrom tensorflow.keras.layers import (Dense, LSTM, Bidirectional, GRU, \n                                     Dropout, Input, LayerNormalization, \n                                     MultiHeadAttention, GlobalAveragePooling1D,\n                                     Conv1D, MaxPooling1D)\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras import mixed_precision\n\n# Configure random seeds for absolute reproducibility in research\nSEED = 42\nos.environ['PYTHONHASHSEED'] = str(SEED)\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\n# Print versions to document the research environment\nprint(\"--- Environment Details ---\")\nprint(f\"TensorFlow Version: {tf.__version__}\")\nprint(f\"Python Version: {sys.version.split()[0]}\")\nprint(f\"NumPy Version: {np.__version__}\")\nprint(f\"Pandas Version: {pd.__version__}\")\nprint(f\"Scikit-Learn Version: {sklearn.__version__}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:51.915540Z","iopub.execute_input":"2026-04-26T03:40:51.915965Z","iopub.status.idle":"2026-04-26T03:40:52.610156Z","shell.execute_reply.started":"2026-04-26T03:40:51.915943Z","shell.execute_reply":"2026-04-26T03:40:52.609307Z"}},"outputs":[],"execution_count":null},{"id":"ddd89a0a","cell_type":"code","source":"\"\"\"\"\ndef seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n\ndef get_strategy():\n\n    IS_TPU = False\n\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        print(\"TPU detected\")\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.TPUStrategy(tpu)\n        IS_TPU = True\n\n    except ValueError:\n\n        gpus = tf.config.list_physical_devices('GPU')\n\n        if len(gpus) > 1:\n            print(f\"{len(gpus)} GPUs detected\")\n            strategy = tf.distribute.MirroredStrategy()\n\n        elif len(gpus) == 1:\n            print(\"Single GPU detected\")\n            strategy = tf.distribute.get_strategy()\n\n        else:\n            print(\"No GPU detected, using CPU\")\n            strategy = tf.distribute.get_strategy()\n\n    AUTO = tf.data.AUTOTUNE\n    REPLICAS = strategy.num_replicas_in_sync\n\n    print(f\"Replicas in sync: {REPLICAS}\")\n\n    return strategy, REPLICAS, IS_TPU\n\n\nseed_everything()\n\nSTRATEGY, N_REPLICAS, Ine_S_TPU = get_strategy()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:52.612029Z","iopub.execute_input":"2026-04-26T03:40:52.612547Z","iopub.status.idle":"2026-04-26T03:40:52.618461Z","shell.execute_reply.started":"2026-04-26T03:40:52.612521Z","shell.execute_reply":"2026-04-26T03:40:52.617706Z"}},"outputs":[],"execution_count":null},{"id":"1f9e0768","cell_type":"code","source":"from pathlib import Path\nimport os\n\nprint(\"Current working directory:\", Path().resolve())\n\nDATA_DIR = Path(\"/kaggle/input/competitions/asl-signs\")\n\nTRAIN_CSV = DATA_DIR / \"train.csv\"\nLANDMARK_DIR = DATA_DIR / \"train_landmark_files\"\nSIGN_MAP = DATA_DIR / \"sign_to_prediction_index_map.json\"\n\nPROJECT_ROOT = Path(\"../\")\n\nEVAL_DIR = PROJECT_ROOT / \"Evaluation_Plots\"\nMODEL_DIR = PROJECT_ROOT / \"Saved_Models\"\nPRED_DIR = PROJECT_ROOT / \"Predictions\"\nHIST_DIR = PROJECT_ROOT / \"Training_Histories\"\n\nprint(\"\\nDataset paths check:\")\nprint(\"DATA_DIR:\", DATA_DIR)\nprint(\"Train CSV exists:\", TRAIN_CSV.exists())\nprint(\"Landmark folder exists:\", LANDMARK_DIR.exists())\nprint(\"Sign map exists:\", SIGN_MAP.exists())\n\nparquet_folders = list(LANDMARK_DIR.glob(\"*\"))\nprint(\"Number of parquet folders:\", len(parquet_folders))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:52.619627Z","iopub.execute_input":"2026-04-26T03:40:52.619930Z","iopub.status.idle":"2026-04-26T03:40:52.652194Z","shell.execute_reply.started":"2026-04-26T03:40:52.619905Z","shell.execute_reply":"2026-04-26T03:40:52.651457Z"}},"outputs":[],"execution_count":null},{"id":"2f479d31","cell_type":"code","source":"train_df = pd.read_csv(TRAIN_CSV)\ndisplay(train_df.head())\ndisplay(train_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:52.652995Z","iopub.execute_input":"2026-04-26T03:40:52.653367Z","iopub.status.idle":"2026-04-26T03:40:52.882173Z","shell.execute_reply.started":"2026-04-26T03:40:52.653344Z","shell.execute_reply":"2026-04-26T03:40:52.881469Z"}},"outputs":[],"execution_count":null},{"id":"80b4d361","cell_type":"code","source":"print(\"Number of samples in train.csv:\", len(train_df))\nprint(\"Number of unique signs:\", train_df[\"sign\"].nunique())\nprint(\"Number of participants:\", train_df[\"participant_id\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:52.883279Z","iopub.execute_input":"2026-04-26T03:40:52.883654Z","iopub.status.idle":"2026-04-26T03:40:52.895285Z","shell.execute_reply.started":"2026-04-26T03:40:52.883627Z","shell.execute_reply":"2026-04-26T03:40:52.894655Z"}},"outputs":[],"execution_count":null},{"id":"89a13c9e","cell_type":"markdown","source":"### Spatial-Temporal Feature Engineering and Preprocessing\nIn this phase, we define the core preprocessing pipeline. Instead of feeding all 543 raw MediaPipe landmarks into the models, we isolate the most informative nodes (Lips, Eyes, Nose, and Hands) to reduce noise and computational complexity. \n\nFurthermore, we implement a custom Keras Layer (`Preprocess`) that performs the following operations directly within the TensorFlow graph:\n1. **NaN Handling:** Computes safe means and standard deviations to normalize coordinates, replacing missing landmarks (NaNs) seamlessly.\n2. **Normalization:** Centers the coordinates based on a reference point.\n3. **Temporal Dynamics (Velocity & Acceleration):** Computes the first derivative (`dx`) and second derivative (`dx2`) of the coordinates across frames to capture motion speed and trajectory.\n4. **Feature Fusion:** Concatenates positions, velocities, and accelerations into a robust feature vector (shape: `[Frames, Channels]`) optimized for sequential models.","metadata":{}},{"id":"b4c8d995","cell_type":"code","source":"# Constants for data dimensions and padding\nROWS_PER_FRAME = 543\nMAX_LEN = 384\nCROP_LEN = MAX_LEN\nNUM_CLASSES  = 250\nPAD = -100.\n\n# ---------------------------------------------------------------------------\n# Feature Selection: Isolating Informative Landmarks (Lips, Nose, Eyes, Hands)\n# ---------------------------------------------------------------------------\nNOSE = [1, 2, 98, 327]\nLNOSE = [98]\nRNOSE = [327]\n\nLIP = [ \n    0, 61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\n\nLLIP = [84, 181, 91, 146, 61, 185, 40, 39, 37, 87, 178, 88, 95, 78, 191, 80, 81, 82]\nRLIP = [314, 405, 321, 375, 291, 409, 270, 269, 267, 317, 402, 318, 324, 308, 415, 310, 311, 312]\n\nPOSE = [500, 502, 504, 501, 503, 505, 512, 513]\nLPOSE = [513, 505, 503, 501]\nRPOSE = [512, 504, 502, 500]\n\nREYE = [\n    33, 7, 163, 144, 145, 153, 154, 155, 133,\n    246, 161, 160, 159, 158, 157, 173,\n]\nLEYE = [\n    263, 249, 390, 373, 374, 380, 381, 382, 362,\n    466, 388, 387, 386, 385, 384, 398,\n]\n\n# MediaPipe Hand Landmarks indices\nLHAND = np.arange(468, 489).tolist()\nRHAND = np.arange(522, 543).tolist()\n\n# Final concatenated feature list\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE\n\nNUM_NODES = len(POINT_LANDMARKS)\n# Channels = (X, Y) * (Position, Velocity, Acceleration) = 2 * 3 = 6 per node\nCHANNELS = 6 * NUM_NODES \n\nprint(f\"Total Selected Nodes: {NUM_NODES}\")\nprint(f\"Total Output Channels per frame: {CHANNELS}\")\n\n# ---------------------------------------------------------------------------\n# Utility Functions for robust mathematical operations\n# ---------------------------------------------------------------------------\ndef tf_nan_mean(x, axis=0, keepdims=False):\n    \"\"\"Computes the mean of a tensor ignoring NaN values.\"\"\"\n    sum_val = tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis, keepdims=keepdims)\n    count_val = tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis, keepdims=keepdims)\n    return sum_val / count_val\n\ndef tf_nan_std(x, center=None, axis=0, keepdims=False):\n    \"\"\"Computes the standard deviation of a tensor ignoring NaN values.\"\"\"\n    if center is None:\n        center = tf_nan_mean(x, axis=axis,  keepdims=True)\n    d = x - center\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis, keepdims=keepdims))\n\n# ---------------------------------------------------------------------------\n# Custom Keras Layer for Spatial-Temporal Feature Engineering\n# ---------------------------------------------------------------------------\nclass Preprocess(tf.keras.layers.Layer):\n    \"\"\"\n    A custom TensorFlow layer that normalizes coordinates, handles NaNs, \n    and computes dynamic temporal features (velocity and acceleration).\n    \"\"\"\n    def __init__(self, max_len=MAX_LEN, point_landmarks=POINT_LANDMARKS, **kwargs):\n        super().__init__(**kwargs)\n        self.max_len = max_len\n        self.point_landmarks = point_landmarks\n\n    def call(self, inputs):\n        if len(inputs.shape) == 3:\n            x = inputs[None, ...]\n        else:\n            x = inputs\n        \n        # Center normalization based on a reference point\n        mean = tf_nan_mean(tf.gather(x, [17], axis=2), axis=[1, 2], keepdims=True)\n        mean = tf.where(tf.math.is_nan(mean), tf.constant(0.5, x.dtype), mean)\n        \n        # Isolate selected landmarks\n        x = tf.gather(x, self.point_landmarks, axis=2) # Shape: N, T, P, C\n        std = tf_nan_std(x, center=mean, axis=[1, 2], keepdims=True)\n        x = (x - mean) / std\n\n        if self.max_len is not None:\n            x = x[:, :self.max_len]\n            \n        length = tf.shape(x)[1]\n        \n        # Retain only X and Y coordinates (drop Z for classification efficiency)\n        x = x[..., :2]\n\n        # Calculate Velocity (First Derivative - dx)\n        dx = tf.cond(\n            tf.shape(x)[1] > 1,\n            lambda: tf.pad(x[:, 1:] - x[:, :-1], [[0, 0], [0, 1], [0, 0], [0, 0]]),\n            lambda: tf.zeros_like(x)\n        )\n\n        # Calculate Acceleration (Second Derivative - dx2)\n        dx2 = tf.cond(\n            tf.shape(x)[1] > 2,\n            lambda: tf.pad(x[:, 2:] - x[:, :-2], [[0, 0], [0, 2], [0, 0], [0, 0]]),\n            lambda: tf.zeros_like(x)\n        )\n\n        # Concatenate Position, Velocity, and Acceleration\n        x = tf.concat([\n            tf.reshape(x, (-1, length, 2 * len(self.point_landmarks))),\n            tf.reshape(dx, (-1, length, 2 * len(self.point_landmarks))),\n            tf.reshape(dx2, (-1, length, 2 * len(self.point_landmarks))),\n        ], axis=-1)\n        \n        # Replace any remaining NaNs with zeros\n        x = tf.where(tf.math.is_nan(x), tf.constant(0., x.dtype), x)\n        \n        return x\n\n    def get_config(self):\n        \"\"\"Required for layer serialization and model saving.\"\"\"\n        config = super().get_config()\n        config.update({\n            \"max_len\": self.max_len,\n            \"point_landmarks\": self.point_landmarks,\n        })\n        return config","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:52.897221Z","iopub.execute_input":"2026-04-26T03:40:52.897529Z","iopub.status.idle":"2026-04-26T03:40:52.917848Z","shell.execute_reply.started":"2026-04-26T03:40:52.897505Z","shell.execute_reply":"2026-04-26T03:40:52.917109Z"}},"outputs":[],"execution_count":null},{"id":"beff1ea0","cell_type":"markdown","source":"### Phase 4: Data Augmentation and Parquet Pipeline Integration\nThis section adapts the standard TFRecord-based data loading pipeline to directly read from `.parquet` files using a Python generator wrapped in `tf.data.Dataset.from_generator`. \n\nIt includes advanced spatial-temporal augmentations specifically designed for sign language recognition:\n1. `flip_lr`: Simulates left-handed vs right-handed signers.\n2. `resample`: Alters the speed of the sign dynamically.\n3. `spatial_random_affine`: Applies rotation, scaling, and shear to simulate different camera angles.\n4. `spatial_mask` & `temporal_mask`: Adds robustness by randomly obscuring parts of the frame or sequence.\n\nFinally, the `get_parquet_dataset` function constructs an optimized, prefetching TensorFlow dataset ready for model training.","metadata":{}},{"id":"20196a9c","cell_type":"code","source":"# 1. Encode Sign Labels to Integers (0 to 249)\nif 'label' not in train_df.columns:\n    sign_list = sorted(train_df['sign'].unique())\n    sign_to_label = {sign: label for label, sign in enumerate(sign_list)}\n    label_to_sign = {label: sign for sign, label in sign_to_label.items()}\n    train_df['label'] = train_df['sign'].map(sign_to_label)\n    print(f\"Encoded {len(sign_list)} unique signs.\")\n\n# Initialize the Preprocess layer defined in the previous cell\npreprocess_layer = Preprocess(max_len=MAX_LEN, point_landmarks=POINT_LANDMARKS)\n\n# ---------------------------------------------------------------------------\n# Core Parquet Reader Function\n# ---------------------------------------------------------------------------\ndef load_parquet_video(file_path):\n    try:\n        df = pd.read_parquet(file_path, columns=['x', 'y', 'z'], engine='pyarrow')\n        coords = df.values.astype(np.float32)\n        frames = len(coords) // ROWS_PER_FRAME\n        return coords.reshape(frames, ROWS_PER_FRAME, 3)\n    except Exception as e:\n        print(f\"Error loading {file_path}: {e}\", file=sys.stderr)\n        return np.zeros((0, ROWS_PER_FRAME, 3), dtype=np.float32)\n\n# ---------------------------------------------------------------------------\n# Data Augmentation Functions (Math-Safe)\n# ---------------------------------------------------------------------------\ndef filter_nans_tf(x, ref_point=POINT_LANDMARKS):\n    mask = tf.math.logical_not(tf.reduce_all(tf.math.is_nan(tf.gather(x, ref_point, axis=1)), axis=[-2, -1]))\n    x = tf.boolean_mask(x, mask, axis=0)\n    return x\n\ndef flip_lr(x):\n    x_coord, y_coord, z_coord = tf.unstack(x, axis=-1)\n    x_coord = 1.0 - x_coord\n    new_x = tf.stack([x_coord, y_coord, z_coord], -1)\n    new_x = tf.transpose(new_x, [1, 0, 2])\n    \n    lhand = tf.gather(new_x, LHAND, axis=0)\n    rhand = tf.gather(new_x, RHAND, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LHAND)[..., None], rhand)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RHAND)[..., None], lhand)\n    \n    llip = tf.gather(new_x, LLIP, axis=0)\n    rlip = tf.gather(new_x, RLIP, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LLIP)[..., None], rlip)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RLIP)[..., None], llip)\n    \n    lpose = tf.gather(new_x, LPOSE, axis=0)\n    rpose = tf.gather(new_x, RPOSE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LPOSE)[..., None], rpose)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RPOSE)[..., None], lpose)\n    \n    leye = tf.gather(new_x, LEYE, axis=0)\n    reye = tf.gather(new_x, REYE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LEYE)[..., None], reye)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(REYE)[..., None], leye)\n    \n    lnose = tf.gather(new_x, LNOSE, axis=0)\n    rnose = tf.gather(new_x, RNOSE, axis=0)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(LNOSE)[..., None], rnose)\n    new_x = tf.tensor_scatter_nd_update(new_x, tf.constant(RNOSE)[..., None], lnose)\n    \n    new_x = tf.transpose(new_x, [1, 0, 2])\n    return new_x\n\ndef interp1d_(x, target_len, method='random'):\n    target_len = tf.maximum(1, target_len)\n    width = tf.shape(x)[1] \n    size = [target_len, width]\n    \n    if method == 'random':\n        rand_val = tf.random.uniform(())\n        if rand_val < 0.33:\n            x = tf.image.resize(x, size, 'bilinear')\n        elif rand_val < 0.66:\n            x = tf.image.resize(x, size, 'bicubic')\n        else:\n            x = tf.image.resize(x, size, 'nearest')\n    else:\n        x = tf.image.resize(x, size, method)\n    return x\n\ndef resample(x, rate=(0.8, 1.2)):\n    rate = tf.random.uniform((), rate[0], rate[1])\n    length = tf.shape(x)[0]\n    new_size = tf.cast(rate * tf.cast(length, tf.float32), tf.int32)\n    new_size = tf.maximum(1, new_size)\n    new_x = interp1d_(x, new_size)\n    return new_x\n\ndef spatial_random_affine(xyz, scale=(0.8, 1.2), shear=(-0.15, 0.15), shift=(-0.1, 0.1), degree=(-30, 30)):\n    center = tf.constant([0.5, 0.5])\n    if scale is not None:\n        scale_val = tf.random.uniform((), *scale)\n        xyz = scale_val * xyz\n\n    if shear is not None:\n        xy = xyz[..., :2]\n        z = xyz[..., 2:]\n        shear_x = tf.random.uniform((), *shear)\n        shear_y = tf.random.uniform((), *shear)\n        if tf.random.uniform(()) < 0.5:\n            shear_x = 0.0\n        else:\n            shear_y = 0.0\n        shear_mat = tf.identity([[1.0, shear_x], [shear_y, 1.0]])\n        xy = xy @ shear_mat\n        center = center + [shear_y, shear_x]\n        xyz = tf.concat([xy, z], axis=-1)\n\n    if degree is not None:\n        xy = xyz[..., :2]\n        z = xyz[..., 2:]\n        xy -= center\n        degree_val = tf.random.uniform((), *degree)\n        radian = degree_val / 180.0 * 3.14159265359\n        c = tf.math.cos(radian)\n        s = tf.math.sin(radian)\n        rotate_mat = tf.identity([[c, s], [-s, c]])\n        xy = xy @ rotate_mat\n        xy = xy + center\n        xyz = tf.concat([xy, z], axis=-1)\n\n    if shift is not None:\n        shift_val = tf.random.uniform((), *shift)\n        xyz = xyz + shift_val\n\n    return xyz\n\ndef temporal_crop(x, length=MAX_LEN):\n    l = tf.shape(x)[0]\n    max_offset = tf.maximum(1, l - length + 1)\n    offset = tf.random.uniform((), 0, max_offset, dtype=tf.int32)\n    x = x[offset:offset + length]\n    return x\n\ndef temporal_mask(x, size=(0.2, 0.4), mask_value=float('nan')):\n    l = tf.shape(x)[0]\n    mask_size = tf.random.uniform((), *size)\n    mask_size = tf.cast(tf.cast(l, tf.float32) * mask_size, tf.int32)\n    mask_size = tf.maximum(1, mask_size)\n    max_offset = tf.maximum(1, l - mask_size + 1)\n    mask_offset = tf.random.uniform((), 0, max_offset, dtype=tf.int32)\n    indices = tf.range(mask_offset, mask_offset + mask_size)[..., None]\n    updates = tf.fill([mask_size, ROWS_PER_FRAME, 3], mask_value)\n    x = tf.tensor_scatter_nd_update(x, indices, updates)\n    return x\n\ndef spatial_mask(x, size=(0.2, 0.4), mask_value=float('nan')):\n    mask_offset_y = tf.random.uniform(())\n    mask_offset_x = tf.random.uniform(())\n    mask_size = tf.random.uniform((), *size)\n    mask_x = (mask_offset_x < x[..., 0]) & (x[..., 0] < mask_offset_x + mask_size)\n    mask_y = (mask_offset_y < x[..., 1]) & (x[..., 1] < mask_offset_y + mask_size)\n    mask = mask_x & mask_y\n    x = tf.where(mask[..., None], mask_value, x)\n    return x\n\ndef augment_fn(x, max_len=None):\n    if tf.random.uniform(()) < 0.8:\n        x = resample(x, (0.5, 1.5))\n    if tf.random.uniform(()) < 0.5:\n        x = flip_lr(x)\n    if max_len is not None:\n        x = temporal_crop(x, max_len)\n    if tf.random.uniform(()) < 0.75:\n        x = spatial_random_affine(x)\n    if tf.random.uniform(()) < 0.5:\n        x = temporal_mask(x)\n    if tf.random.uniform(()) < 0.5:\n        x = spatial_mask(x)\n    return x\n\n# ---------------------------------------------------------------------------\n# TensorFlow Data Pipeline Implementation\n# ---------------------------------------------------------------------------\ndef process_data(coord, label, augment=False, max_len=MAX_LEN):\n    coord = filter_nans_tf(coord)\n    if augment:\n        coord = augment_fn(coord, max_len=max_len)\n    \n    coord = tf.ensure_shape(coord, (None, ROWS_PER_FRAME, 3))\n    \n    processed = preprocess_layer(coord)\n    processed = tf.squeeze(processed, axis=0) \n    \n    processed = tf.where(tf.math.is_nan(processed), tf.zeros_like(processed), processed)\n    processed = tf.where(tf.math.is_inf(processed), tf.zeros_like(processed), processed)\n    \n    processed = tf.cast(processed, tf.float32)\n    one_hot_label = tf.one_hot(label, NUM_CLASSES)\n    \n    return processed, one_hot_label\n\ndef get_parquet_dataset(df, data_dir=DATA_DIR, batch_size=64, max_len=MAX_LEN, augment=False, shuffle=False):\n    paths = df['path'].astype(str).values\n    labels = df['label'].values.astype(np.int32)\n    \n    ds = tf.data.Dataset.from_tensor_slices((paths, labels))\n    \n    if shuffle:\n        ds = ds.shuffle(len(df), reshuffle_each_iteration=True)\n        \n    def py_load_video(path_val):\n        path_str = path_val.decode('utf-8')\n        file_path = os.path.join(str(data_dir), path_str.replace('\\\\', '/'))\n        file_path = os.path.normpath(file_path)\n        coords = load_parquet_video(file_path)\n        return coords\n\n    def load_video_tf(path_tensor, label_tensor):\n        coords = tf.numpy_function(py_load_video, [path_tensor], tf.float32)\n        coords.set_shape((None, ROWS_PER_FRAME, 3))\n        return coords, label_tensor\n\n    ds = ds.map(load_video_tf, num_parallel_calls=tf.data.AUTOTUNE)\n    \n    ds = ds.filter(lambda x, y: tf.shape(x)[0] > 0)\n    \n    ds = ds.map(lambda x, y: process_data(x, y, augment=augment, max_len=max_len), \n                num_parallel_calls=tf.data.AUTOTUNE)\n    \n    ds = ds.padded_batch(\n        batch_size, \n        padding_values=(tf.cast(PAD, tf.float32), tf.cast(0.0, tf.float32)), \n        padded_shapes=([max_len, CHANNELS], [NUM_CLASSES]), \n        drop_remainder=True\n    )\n    \n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    return ds\n\n# ---------------------------------------------------------------------------\n# Pipeline Sanity Check\n# ---------------------------------------------------------------------------\nprint(\"Testing the Parquet Pipeline...\")\ntest_df_subset = train_df.head(10) \ntest_ds = get_parquet_dataset(test_df_subset, batch_size=2, augment=True)\n\nfor batch_x, batch_y in test_ds.take(1):\n    print(f\"Batch X Shape: {batch_x.shape}\")\n    print(f\"Batch Y Shape: {batch_y.shape}\")\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:52.918675Z","iopub.execute_input":"2026-04-26T03:40:52.918949Z","iopub.status.idle":"2026-04-26T03:40:55.263751Z","shell.execute_reply.started":"2026-04-26T03:40:52.918927Z","shell.execute_reply":"2026-04-26T03:40:55.262930Z"}},"outputs":[],"execution_count":null},{"id":"f140e97a","cell_type":"markdown","source":"###  Visualizing MediaPipe Landmarks \nBefore feeding the data into complex neural networks, it is essential to visually verify the integrity of the spatial coordinates. The following code iterates through the parquet files, locates a sequence with valid hand landmarks (filtering out missing frames), and generates an interactive 2D animation of the hand skeletal connections over time. This confirms that the coordinate extraction and reshaping processes are correct.","metadata":{}},{"id":"0afdd5d1","cell_type":"code","source":"from IPython.display import HTML\nimport matplotlib.pyplot as plt\nfrom matplotlib.animation import FuncAnimation\nimport numpy as np\nimport os\n\"\"\"\n\n# ---------------------------------------------------------\n# MediaPipe Hand Connections\n# Defines skeletal edges between the 21 hand landmarks\n# ---------------------------------------------------------\n\nHAND_EDGES = [\n    (0,1),(1,2),(2,3),(3,4),\n    (0,5),(5,6),(6,7),(7,8),\n    (5,9),(9,10),(10,11),(11,12),\n    (9,13),(13,14),(14,15),(15,16),\n    (13,17),(0,17),(17,18),(18,19),(19,20)\n]\n\n\n# ---------------------------------------------------------\n# Utility: Filter frames that contain only NaN coordinates\n# Some sequences contain missing frames that should be removed\n# ---------------------------------------------------------\n\ndef filter_nans(frames):\n    mask = ~np.isnan(frames).all(axis=(-2,-1))\n    return frames[mask]\n\n\n# ---------------------------------------------------------\n# Locate a valid sequence in the dataset\n# The sequence must contain a visible hand motion\n# ---------------------------------------------------------\n\nsample_frames = None\n\nprint(\"Searching for a valid sequence\")\n\nfor row in train_df.itertuples():\n\n    file_path = os.path.join(DATA_DIR, str(row.path).replace(\"\\\\\",\"/\"))\n    file_path = os.path.normpath(file_path)\n\n    coords = load_parquet_video(file_path)\n\n    if coords.shape[0] == 0:\n        continue\n\n    lhand = coords[:,LHAND,:]\n\n    valid_frames = filter_nans(lhand)\n\n    if len(valid_frames) > 20:\n        sample_frames = coords\n        print(\"Sequence found:\", row.sign)\n        break\n\n\n# ---------------------------------------------------------\n# Core Animation Function\n# Draws landmarks and skeleton edges frame-by-frame\n# ---------------------------------------------------------\n\ndef animate_frames(frames, edges=None, idxs=None):\n\n    frames = filter_nans(frames)\n\n    fig, ax = plt.subplots(figsize=(6,6))\n\n    def plot_frame(i):\n\n        ax.clear()\n\n        frame = np.nan_to_num(frames[i])\n\n        x = frame[:,0]\n        y = frame[:,1]\n\n        ax.scatter(x,y,color=\"dodgerblue\",s=40)\n\n        if idxs is not None:\n            for j in range(len(x)):\n                ax.text(x[j],y[j],str(idxs[j]),fontsize=7)\n\n        if edges is not None:\n            for e in edges:\n                ax.plot(\n                    [x[e[0]],x[e[1]]],\n                    [y[e[0]],y[e[1]]],\n                    color=\"salmon\",\n                    linewidth=2\n                )\n\n        ax.invert_yaxis()\n\n        ax.set_xticks([])\n        ax.set_yticks([])\n\n    anim = FuncAnimation(fig, plot_frame, frames=len(frames), interval=100)\n\n    plt.close(fig)\n\n    return HTML(anim.to_jshtml())\n\n\n# ---------------------------------------------------------\n# Save Animation Function\n# Used to export animations for research paper figures\n# ---------------------------------------------------------\n\ndef save_animation(frames, filename, edges=None, idxs=None):\n\n    frames = filter_nans(frames)\n\n    fig, ax = plt.subplots(figsize=(6,6))\n\n    def plot_frame(i):\n\n        ax.clear()\n\n        frame = np.nan_to_num(frames[i])\n\n        x = frame[:,0]\n        y = frame[:,1]\n\n        ax.scatter(x,y,color=\"dodgerblue\",s=40)\n\n        if idxs is not None:\n            for j in range(len(x)):\n                ax.text(x[j],y[j],str(idxs[j]),fontsize=7)\n\n        if edges is not None:\n            for e in edges:\n                ax.plot([x[e[0]],x[e[1]]],[y[e[0]],y[e[1]]],color=\"salmon\")\n\n        ax.invert_yaxis()\n\n        ax.set_xticks([])\n        ax.set_yticks([])\n\n    anim = FuncAnimation(fig, plot_frame, frames=len(frames), interval=100)\n\n    save_path = os.path.join(\"Research Paper\",\"Evaluation_Plots\",filename)\n\n    anim.save(save_path, writer=\"pillow\", fps=10)\n\n    plt.close(fig)\n\n    print(\"Animation saved to:\", save_path)\n\n\n# ---------------------------------------------------------\n# Display Left Hand Landmarks\n# ---------------------------------------------------------\n\nprint(\"Left Hand Motion\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,LHAND],\n        edges=HAND_EDGES,\n        idxs=list(range(len(LHAND)))\n    )\n)\n\n\n# ---------------------------------------------------------\n# Display Right Hand Landmarks\n# ---------------------------------------------------------\n\nprint(\"Right Hand Motion\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,RHAND],\n        edges=HAND_EDGES,\n        idxs=list(range(len(RHAND)))\n    )\n)\n\n\n# ---------------------------------------------------------\n# Display Face Landmarks\n# ---------------------------------------------------------\n\nprint(\"Face Landmarks\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,LIP + LEYE + REYE + NOSE],\n        idxs=LIP + LEYE + REYE + NOSE\n    )\n)\n\n\n# ---------------------------------------------------------\n# Display All Selected Landmarks Used by the Model\n# ---------------------------------------------------------\n\nprint(\"Full Landmark Representation\")\n\ndisplay(\n    animate_frames(\n        sample_frames[:,POINT_LANDMARKS],\n        idxs=POINT_LANDMARKS\n    )\n)\n\n\n# ---------------------------------------------------------\n# Example of Augmented Sequence Visualization\n# ---------------------------------------------------------\n\nprint(\"Augmented Sequence Example\")\n\naugmented = augment_fn(sample_frames, max_len=MAX_LEN).numpy()\n\ndisplay(\n    animate_frames(\n        augmented[:,POINT_LANDMARKS],\n        idxs=POINT_LANDMARKS\n    )\n)\n\n\n# ---------------------------------------------------------\n# Save animations for later use\n# ---------------------------------------------------------\n\nsave_animation(sample_frames[:,RHAND],\"right_hand.gif\",edges=HAND_EDGES,idxs=list(range(len(RHAND))))\nsave_animation(sample_frames[:,POINT_LANDMARKS],\"full_landmarks.gif\",idxs=POINT_LANDMARKS)\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:55.266320Z","iopub.execute_input":"2026-04-26T03:40:55.266670Z","iopub.status.idle":"2026-04-26T03:40:55.292500Z","shell.execute_reply.started":"2026-04-26T03:40:55.266643Z","shell.execute_reply":"2026-04-26T03:40:55.291879Z"}},"outputs":[],"execution_count":null},{"id":"0896fe58","cell_type":"markdown","source":"###  Final Data Shape and Tensor Inspection\nBefore defining the neural network architectures, we extract a single batch from our parquet dataset generator to inspect the exact tensor dimensions and the statistical properties of the engineered features. This confirms that the normalization, padding, and one-hot encoding have been applied correctly.","metadata":{}},{"id":"fc7e6210","cell_type":"code","source":"import numpy as np\n\nprint(\"Fetching a single batch from the dataset...\")\n# We use the test_ds created in the previous cell\nfor batch_x, batch_y in test_ds.take(1):\n    \n    x_numpy = batch_x.numpy()\n    y_numpy = batch_y.numpy()\n    \n    print(\"\\n--- 1. Input Features (X) ---\")\n    print(f\"Shape: {x_numpy.shape} -> (Batch, Frames, Channels)\")\n    print(f\"Data Type: {x_numpy.dtype}\")\n    # Check normalization properties (should be centered around 0)\n    print(f\"Global Min Value: {np.min(x_numpy):.4f}\")\n    print(f\"Global Max Value: {np.max(x_numpy):.4f}\")\n    print(f\"Global Mean: {np.mean(x_numpy):.4f}\")\n    \n    print(\"\\n--- 2. Output Labels (Y) ---\")\n    print(f\"Shape: {y_numpy.shape} -> (Batch, Num_Classes)\")\n    print(f\"Data Type: {y_numpy.dtype}\")\n    \n    # Show the active class for the first sample in the batch\n    first_sample_label_index = np.argmax(y_numpy[0])\n    # Assuming label_to_sign dictionary is available from previous cell\n    sign_word = label_to_sign.get(first_sample_label_index, \"Unknown\")\n    print(f\"First Sample One-Hot Label Index: {first_sample_label_index}\")\n    print(f\"Corresponding Sign Word: '{sign_word}'\")\n    \n    print(\"\\n Data is perfectly shaped and ready for modeling!\")\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:55.293379Z","iopub.execute_input":"2026-04-26T03:40:55.294065Z","iopub.status.idle":"2026-04-26T03:40:55.571243Z","shell.execute_reply.started":"2026-04-26T03:40:55.294031Z","shell.execute_reply":"2026-04-26T03:40:55.570540Z"}},"outputs":[],"execution_count":null},{"id":"91dcbfeb","cell_type":"markdown","source":"### Phase 2: Stratified Data Splitting and Dataset Creation\nTo ensure robust model evaluation and prevent data leakage, we split the dataset into Training (80%), Validation (10%), and Testing (10%) sets. \n\nKey considerations for this research pipeline:\n1. **Stratification:** We stratify the split based on the target labels to maintain a consistent class distribution across all splits (crucial for the 250-class ISLR dataset).\n2. **Reproducibility:** A fixed random seed ensures identical splits across different execution environments.\n3. **Artifact Saving:** The splits are saved as CSV files. This guarantees that all subsequent baseline and advanced models, as well as the final Ensemble, are evaluated on the exact same unseen test instances.\n4. **Optimized Pipelines:** We instantiate `tf.data` pipelines with `AUTOTUNE` prefetching. Augmentation and shuffling are strictly applied only to the training set.","metadata":{}},{"id":"0dac7a1e","cell_type":"code","source":"\nfrom sklearn.model_selection import train_test_split\n\n\nprint(\"Initiating Stratified Data Splitting...\")\n\n# 1. Create data directory if it doesn't exist (from your project structure)\nos.makedirs(\"data\", exist_ok=True)\n\n\ntrain_df_split, temp_df = train_test_split(\n    train_df, \n    test_size=0.20, \n    random_state=SEED, \n    stratify=train_df['label']\n)\n\n# Second split: Split the 40% Temporary equally into 20% Validation and 20% Test\nval_df_split, test_df_split = train_test_split(\n    temp_df, \n    test_size=0.50, \n    random_state=SEED, \n    stratify=temp_df['label']\n)\n\n# 3. Save the splits to disk for absolute reproducibility during Ensemble\ntrain_split_path = os.path.join(\"data\", \"train_split.csv\")\nval_split_path = os.path.join(\"data\", \"val_split.csv\")\ntest_split_path = os.path.join(\"data\", \"test_split.csv\")\n\ntrain_df_split.to_csv(train_split_path, index=False)\nval_df_split.to_csv(val_split_path, index=False)\ntest_df_split.to_csv(test_split_path, index=False)\n\nprint(f\"Data Splitting Complete and Saved to 'data/' directory.\")\nprint(f\"Total Samples: {len(train_df)}\")\nprint(f\"--> Training Set:   {len(train_df_split)} samples ({len(train_df_split)/len(train_df)*100:.1f}%)\")\nprint(f\"--> Validation Set: {len(val_df_split)} samples ({len(val_df_split)/len(train_df)*100:.1f}%)\")\nprint(f\"--> Testing Set:    {len(test_df_split)} samples ({len(test_df_split)/len(train_df)*100:.1f}%)\")\n\n# 4. Create Highly Optimized tf.data.Datasets\nprint(\"\\nConstructing TensorFlow Datasets...\")\n\n# Hyperparameters for training\nBATCH_SIZE = 128 # Adjust this depending on your GPU RAM (e.g., 32 if OOM error occurs, 128 if plenty of VRAM)\n\n# Train Dataset: Needs Augmentation and Shuffling\ntrain_dataset = get_parquet_dataset(\n    train_df_split, \n    data_dir=DATA_DIR, \n    batch_size=BATCH_SIZE, \n    max_len=MAX_LEN, \n    augment=True, \n    shuffle=True\n)\n\n# Validation Dataset: NO Augmentation, NO Shuffling (for accurate metric tracking)\nval_dataset = get_parquet_dataset(\n    val_df_split, \n    data_dir=DATA_DIR, \n    batch_size=BATCH_SIZE, \n    max_len=MAX_LEN, \n    augment=False, \n    shuffle=False\n)\n\n# Test Dataset: NO Augmentation, NO Shuffling (for final paper evaluation)\ntest_dataset = get_parquet_dataset(\n    test_df_split, \n    data_dir=DATA_DIR, \n    batch_size=BATCH_SIZE, \n    max_len=MAX_LEN, \n    augment=False, \n    shuffle=False\n)\n\nprint(\"\\n--- TF Dataset Specifications ---\")\nprint(f\"Train Dataset: {train_dataset}\")\nprint(f\"Val Dataset:   {val_dataset}\")\nprint(f\"Test Dataset:  {test_dataset}\")\nprint(\" Data Pipelines are heavily optimized and ready for model consumption!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:55.572129Z","iopub.execute_input":"2026-04-26T03:40:55.572462Z","iopub.status.idle":"2026-04-26T03:40:58.610538Z","shell.execute_reply.started":"2026-04-26T03:40:55.572395Z","shell.execute_reply":"2026-04-26T03:40:58.609650Z"}},"outputs":[],"execution_count":null},{"id":"ba26a933","cell_type":"markdown","source":"### Test Data And Fixing early Stoping","metadata":{}},{"id":"95cf1199","cell_type":"code","source":"import tensorflow as tf\n\nprint(\"--- Real Dataset Diagnostic Check ---\")\n\n# سحب باتش واحد من بيانات الاختبار الحقيقية\nfor x_v, y_v in val_dataset.take(1):\n    print(f'Validation X Shape: {x_v.shape}')\n    print(f'Validation Labels Shape: {y_v.shape}')\n    print(f'Contains NaNs in Val X: {tf.reduce_any(tf.math.is_nan(x_v)).numpy()}')\n    print(f'Contains NaNs in Val Y: {tf.reduce_any(tf.math.is_nan(y_v)).numpy()}')\n\n# سحب باتش واحد من بيانات التدريب للمقارنة\nfor x_t, y_t in train_dataset.take(1):\n    print(f'Training Labels Shape: {y_t.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:40:58.611673Z","iopub.execute_input":"2026-04-26T03:40:58.611989Z","iopub.status.idle":"2026-04-26T03:41:00.958298Z","shell.execute_reply.started":"2026-04-26T03:40:58.611955Z","shell.execute_reply":"2026-04-26T03:41:00.957354Z"}},"outputs":[],"execution_count":null},{"id":"591f8f52","cell_type":"code","source":"import tensorflow as tf\nfrom tqdm.autonotebook import tqdm\n\nprint(\"--- 🕵️‍♂️ Initiating FULL Validation Dataset Scan ---\")\nnan_found = False\nbatch_num = 0\n\nfor x_v, y_v in tqdm(val_dataset, desc=\"Scanning Val Data\"):\n    batch_num += 1\n    \n    # فحص الـ Inputs\n    if tf.reduce_any(tf.math.is_nan(x_v)).numpy():\n        print(f\"\\n🚨 BINGO! Poisoned Data (NaN) found in Inputs (X) at batch {batch_num}\")\n        nan_found = True\n        break\n        \n    # فحص الـ Labels\n    if tf.reduce_any(tf.math.is_nan(y_v)).numpy():\n        print(f\"\\n🚨 BINGO! Poisoned Data (NaN) found in Labels (Y) at batch {batch_num}\")\n        nan_found = True\n        break\n\nif not nan_found:\n    print(\"\\n✅ 100% CLEAN: The ENTIRE validation dataset is completely free of NaNs!\")\n    print(\"🔥 Conclusion: The bug is inside the Model's Architecture during Inference (training=False)!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:00.959378Z","iopub.execute_input":"2026-04-26T03:41:00.959739Z","iopub.status.idle":"2026-04-26T03:41:52.687130Z","shell.execute_reply.started":"2026-04-26T03:41:00.959714Z","shell.execute_reply":"2026-04-26T03:41:52.686385Z"}},"outputs":[],"execution_count":null},{"id":"9c9442d0","cell_type":"markdown","source":"### Dataset Summary and Pre-Training Report\nBefore initializing the training phase, we generate a comprehensive statistical report of the engineered dataset. This step verifies the integrity of the stratified split and the exact tensor dimensions. A textual summary is exported to the `results` directory, and a visual representation of the data distribution is saved to the `Evaluation_Plots` directory for inclusion in the research methodology section.","metadata":{}},{"id":"ededcd9f","cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\n\n# ── 1. Compile statistics ──────────────────────────────────────────────────\ntotal_samples      = len(train_df)\ntrain_samples      = len(train_df_split)\nval_samples        = len(val_df_split)\ntest_samples       = len(test_df_split)\n\ntrain_class_counts = train_df_split['label'].value_counts()\nval_class_counts   = val_df_split['label'].value_counts()\ntest_class_counts  = test_df_split['label'].value_counts()\n\n# ── 2. Text report ─────────────────────────────────────────────────────────\nreport_text = f\"\"\"\n=========================================================\n          WESSAL PROJECT: DATASET SUMMARY REPORT\n=========================================================\n\n1. GLOBAL DATASET METRICS\n---------------------------------------------------------\n  Total Video Sequences         : {total_samples:,}\n  Total Unique Sign Classes     : {NUM_CLASSES}\n  Landmarks per Frame           : {NUM_NODES} nodes\n  Feature Channels              : {CHANNELS}  (X, Y, dx, dy, dx2, dy2)\n  Maximum Sequence Length       : {MAX_LEN} frames\n  Input Tensor Shape            : (Batch, {MAX_LEN}, {CHANNELS})\n\n2. STRATIFIED DATA SPLIT  (80 / 10 / 10)\n---------------------------------------------------------\n  Training   (80%)              : {train_samples:,} samples\n  Validation (10%)              : {val_samples:,} samples\n  Testing    (10%)              : {test_samples:,} samples\n\n3. CLASS BALANCE  (samples per class)\n---------------------------------------------------------\n  Split        Max    Min    Mean\n  Training     {train_class_counts.max():<6} {train_class_counts.min():<6} {train_class_counts.mean():.1f}\n  Validation   {val_class_counts.max():<6} {val_class_counts.min():<6} {val_class_counts.mean():.1f}\n  Testing      {test_class_counts.max():<6} {test_class_counts.min():<6} {test_class_counts.mean():.1f}\n\n=========================================================\n\"\"\"\n\nprint(report_text)\n\nos.makedirs(\"../results\", exist_ok=True)\nreport_path = \"../results/Pre_Training_Dataset_Report.txt\"\nwith open(report_path, \"w\") as f:\n    f.write(report_text)\nprint(f\"Report saved to: {report_path}\")\n\n# ── 3. Visual report ───────────────────────────────────────────────────────\nCOLORS = [\"#4285F4\", \"#34A853\", \"#FBBC05\"]\nSPLITS = [\"Train\", \"Validation\", \"Test\"]\nSIZES  = [train_samples, val_samples, test_samples]\nMEANS  = [train_class_counts.mean(), val_class_counts.mean(), test_class_counts.mean()]\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 6))\nfig.suptitle(\"WESSAL — Dataset Distribution\", fontsize=15, fontweight=\"bold\", y=1.01)\n\n# Pie chart\nax1.pie(\n    SIZES,\n    labels   = [f\"{s}\\n({v/total_samples*100:.0f}%)\" for s, v in zip(SPLITS, SIZES)],\n    colors   = COLORS,\n    explode  = (0.05, 0, 0),\n    autopct  = \"%1.1f%%\",\n    startangle = 90,\n    textprops  = {\"fontsize\": 11},\n)\nax1.set_title(\"Stratified Split (80 / 10 / 10)\", fontsize=13, fontweight=\"bold\", pad=12)\nax1.axis(\"equal\")\n\n# Bar chart\nbars = ax2.bar(SPLITS, MEANS, color=COLORS, width=0.45, edgecolor=\"white\", linewidth=0.8)\nax2.set_ylabel(\"Mean Sequences per Class\", fontsize=11)\nax2.set_title(\"Average Class Representation per Split\", fontsize=13, fontweight=\"bold\", pad=12)\nax2.spines[[\"top\", \"right\"]].set_visible(False)\nfor bar, val in zip(bars, MEANS):\n    ax2.text(bar.get_x() + bar.get_width() / 2, bar.get_height() + 0.5,\n             f\"{val:.1f}\", ha=\"center\", va=\"bottom\", fontsize=11, fontweight=\"bold\")\n\nplt.tight_layout()\n\nos.makedirs(\"../Evaluation_Plots\", exist_ok=True)\nplot_path = \"../Evaluation_Plots/Data_Split_Distribution.png\"\nplt.savefig(plot_path, dpi=300, bbox_inches=\"tight\")\nprint(f\"Plot saved to: {plot_path}\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:52.688113Z","iopub.execute_input":"2026-04-26T03:41:52.688319Z","iopub.status.idle":"2026-04-26T03:41:53.654226Z","shell.execute_reply.started":"2026-04-26T03:41:52.688298Z","shell.execute_reply":"2026-04-26T03:41:53.653362Z"}},"outputs":[],"execution_count":null},{"id":"cd4453f4-21ac-41d5-ac7d-e801b2b1243d","cell_type":"code","source":"import tensorflow as tf\n\n\nclass SmartMasking(tf.keras.layers.Layer):\n    def __init__(self, pad_value=-100.0, **kwargs):\n        super().__init__(**kwargs)\n        self.pad_value = pad_value\n        self.supports_masking = True\n\n    def compute_mask(self, inputs, mask=None):\n        is_pad = tf.keras.ops.equal(inputs, self.pad_value)\n        is_nan = tf.keras.ops.isnan(inputs)\n        invalid = tf.keras.ops.logical_or(is_pad, is_nan)\n        return tf.keras.ops.logical_not(tf.keras.ops.all(invalid, axis=-1))\n\n    def call(self, inputs):\n        zeros = tf.keras.ops.zeros_like(inputs)\n        return tf.keras.ops.where(tf.keras.ops.isnan(inputs), zeros, inputs)\n\n\nclass ECA(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=5, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.kernel_size = kernel_size\n        self.conv = tf.keras.layers.Conv1D(1, kernel_size=kernel_size, strides=1, padding=\"same\", use_bias=False)\n\n    def call(self, inputs, mask=None):\n        nn = tf.keras.layers.GlobalAveragePooling1D()(inputs, mask=mask)\n        nn = tf.expand_dims(nn, -1)\n        nn = self.conv(nn)\n        nn = tf.squeeze(nn, -1)\n        nn = tf.nn.sigmoid(nn)\n        nn = nn[:, None, :]\n        return inputs * nn\n\n\nclass LateDropout(tf.keras.layers.Layer):\n    def __init__(self, rate, noise_shape=None, start_step=0, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.rate = rate\n        self.start_step = start_step\n        self.dropout = tf.keras.layers.Dropout(rate, noise_shape=noise_shape)\n\n    def build(self, input_shape):\n        super().build(input_shape)\n        agg = tf.VariableAggregation.ONLY_FIRST_REPLICA\n        self._train_counter = tf.Variable(0, dtype=\"int64\", aggregation=agg, trainable=False)\n\n    def call(self, inputs, training=False):\n        x = tf.cond(\n            self._train_counter < self.start_step,\n            lambda: inputs,\n            lambda: self.dropout(inputs, training=training)\n        )\n        if training:\n            self._train_counter.assign_add(1)\n        return x\n\n\nclass CausalDWConv1D(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=17, dilation_rate=1, use_bias=False,\n                 depthwise_initializer='glorot_uniform', name='', **kwargs):\n        super().__init__(name=name, **kwargs)\n        self.causal_pad = tf.keras.layers.ZeroPadding1D(\n            (dilation_rate * (kernel_size - 1), 0), name=name + '_pad'\n        )\n        self.dw_conv = tf.keras.layers.DepthwiseConv1D(\n            kernel_size,\n            strides=1,\n            dilation_rate=dilation_rate,\n            padding='valid',\n            use_bias=use_bias,\n            depthwise_initializer=depthwise_initializer,\n            name=name + '_dwconv'\n        )\n        self.supports_masking = True\n\n    def compute_mask(self, inputs, mask=None):\n        return mask\n\n    def call(self, inputs):\n        x = self.causal_pad(inputs)\n        x = self.dw_conv(x)\n        return x\n\n\nclass MultiHeadSelfAttention(tf.keras.layers.Layer):\n    def __init__(self, dim=256, num_heads=4, dropout=0, **kwargs):\n        super().__init__(**kwargs)\n        self.dim = dim\n        self.scale = self.dim ** -0.5\n        self.num_heads = num_heads\n        self.qkv = tf.keras.layers.Dense(3 * dim, use_bias=False)\n        self.drop1 = tf.keras.layers.Dropout(dropout)\n        self.proj = tf.keras.layers.Dense(dim, use_bias=False)\n        self.supports_masking = True\n\n    def call(self, inputs, mask=None):\n        B = tf.shape(inputs)[0]\n        S = tf.shape(inputs)[1]\n        head_dim = self.dim // self.num_heads\n\n        qkv = self.qkv(inputs)\n        qkv = tf.reshape(qkv, (B, S, self.num_heads, 3 * head_dim))\n        qkv = tf.transpose(qkv, (0, 2, 1, 3))\n        q, k, v = tf.split(qkv, 3, axis=-1)\n\n        attn = tf.matmul(q, k, transpose_b=True) * self.scale\n\n        if mask is not None:\n            attn += (1.0 - tf.cast(mask[:, None, None, :], attn.dtype)) * -1e9\n\n        attn = tf.nn.softmax(attn, axis=-1)\n        attn = self.drop1(attn)\n\n        x = attn @ v\n        x = tf.transpose(x, (0, 2, 1, 3))\n        x = tf.reshape(x, (B, S, self.dim))\n        x = self.proj(x)\n        return x\n\n\ndef Conv1DBlock(channel_size, kernel_size, dilation_rate=1, drop_rate=0.0,\n                expand_ratio=2, se_ratio=0.25, activation='swish', name=None):\n    if name is None:\n        name = str(tf.keras.backend.get_uid(\"mbblock\"))\n\n    def apply(inputs):\n        channels_in = tf.keras.backend.int_shape(inputs)[-1]\n        channels_expand = channels_in * expand_ratio\n\n        skip = inputs\n\n        x = tf.keras.layers.Dense(\n            channels_expand, use_bias=True, activation=activation,\n            name=name + '_expand_conv'\n        )(inputs)\n\n        x = CausalDWConv1D(\n            kernel_size, dilation_rate=dilation_rate,\n            use_bias=False, name=name + '_dwconv'\n        )(x)\n\n        x = tf.keras.layers.BatchNormalization(momentum=0.95, name=name + '_bn')(x)\n        x = ECA()(x)\n\n        x = tf.keras.layers.Dense(\n            channel_size, use_bias=True, name=name + '_project_conv'\n        )(x)\n\n        if drop_rate > 0:\n            x = tf.keras.layers.Dropout(\n                drop_rate, noise_shape=(None, 1, 1), name=name + '_drop'\n            )(x)\n\n        if channels_in == channel_size:\n            x = tf.keras.layers.add([x, skip], name=name + '_add')\n\n        return x\n\n    return apply\n\n\ndef TransformerBlock(dim=256, num_heads=4, expand=4, attn_dropout=0.2,\n                     drop_rate=0.2, activation='swish'):\n    def apply(inputs):\n        x = tf.keras.layers.LayerNormalization(epsilon=1e-6)(inputs)\n        x = MultiHeadSelfAttention(dim=dim, num_heads=num_heads, dropout=attn_dropout)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None, 1, 1))(x)\n        x = tf.keras.layers.Add()([inputs, x])\n        attn_out = x\n\n        x = tf.keras.layers.LayerNormalization(epsilon=1e-6)(x)\n        x = tf.keras.layers.Dense(dim * expand, use_bias=False, activation=activation)(x)\n        x = tf.keras.layers.Dense(dim, use_bias=False)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None, 1, 1))(x)\n        x = tf.keras.layers.Add()([attn_out, x])\n        return x\n\n    return apply\n\n\nclass SignLanguageTransformer:\n    def __init__(self, max_len=384, channels=708, num_classes=250, dim=192,\n                 pad_value=-100.0, dropout_step=0):\n        self.max_len = max_len\n        self.channels = channels\n        self.num_classes = num_classes\n        self.dim = dim\n        self.pad_value = pad_value\n        self.dropout_step = dropout_step\n\n    def build_model(self):\n        inp = tf.keras.Input((self.max_len, self.channels))\n        x = SmartMasking(pad_value=self.pad_value)(inp)\n        ksize = 17\n\n        x = tf.keras.layers.Dense(self.dim, use_bias=False, name='stem_conv')(x)\n        x = tf.keras.layers.BatchNormalization(momentum=0.95, name='stem_bn')(x)\n\n        x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n        x = TransformerBlock(self.dim, expand=2)(x)\n\n        x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n        x = TransformerBlock(self.dim, expand=2)(x)\n\n        if self.dim >= 384:\n            x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n            x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n            x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n            x = TransformerBlock(self.dim, expand=2)(x)\n\n            x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n            x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n            x = Conv1DBlock(self.dim, ksize, drop_rate=0.2)(x)\n            x = TransformerBlock(self.dim, expand=2)(x)\n\n        x = tf.keras.layers.Dense(self.dim * 2, activation=None, name='top_conv')(x)\n        x = tf.keras.layers.GlobalAveragePooling1D()(x)\n        x = LateDropout(0.8, start_step=self.dropout_step)(x)\n        x = tf.keras.layers.Dense(self.num_classes, name='classifier')(x)\n\n        return tf.keras.Model(inp, x, name=\"Transformer_Model_ASL\")\n\n\ndef get_transformer_model(input_shape=(384, 708), num_classes=250):\n    transformer = SignLanguageTransformer(\n        max_len=input_shape[0],\n        channels=input_shape[1],\n        num_classes=num_classes\n    )\n    return transformer.build_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:53.655338Z","iopub.execute_input":"2026-04-26T03:41:53.655676Z","iopub.status.idle":"2026-04-26T03:41:53.683962Z","shell.execute_reply.started":"2026-04-26T03:41:53.655651Z","shell.execute_reply":"2026-04-26T03:41:53.683185Z"}},"outputs":[],"execution_count":null},{"id":"10138011-01db-4ecc-aa97-79b7dcd32403","cell_type":"code","source":"model = get_transformer_model(input_shape=(MAX_LEN, CHANNELS), num_classes=NUM_CLASSES)\nmodel_name = model.name","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:53.684804Z","iopub.execute_input":"2026-04-26T03:41:53.685080Z","iopub.status.idle":"2026-04-26T03:41:55.109038Z","shell.execute_reply.started":"2026-04-26T03:41:53.685056Z","shell.execute_reply":"2026-04-26T03:41:55.108434Z"}},"outputs":[],"execution_count":null},{"id":"e0574db0-1c52-462b-9173-afa57f7453cd","cell_type":"code","source":"\n# 2. Load the exact matching weights\nmodel_path = '/kaggle/input/models/hassanabdulrazeq/trans/tensorflow2/default/1/Transformer.weights.h5'\n\nif os.path.exists(model_path):\n    print(\"Loading weights from:\", model_path)\n    try:\n        model.load_weights(model_path)\n        print(\"Weights loaded successfully with EXACT match.\")\n    except ValueError as e:\n        print(f\"Mismatch Error: {e}\")\n        print(\"Fallback: Loading matching weights only...\")\n        model.load_weights(model_path, by_name=True, skip_mismatch=True)\nelse:\n    print(\"Warning: Weights file not found. Check the path.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:55.110041Z","iopub.execute_input":"2026-04-26T03:41:55.110563Z","iopub.status.idle":"2026-04-26T03:41:55.463195Z","shell.execute_reply.started":"2026-04-26T03:41:55.110536Z","shell.execute_reply":"2026-04-26T03:41:55.462493Z"}},"outputs":[],"execution_count":null},{"id":"25ea9ae2","cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import metrics, losses, optimizers\n\n# 1. Setup Steps\nsteps_per_epoch  = len(train_df_split) // BATCH_SIZE\ntotal_steps      = steps_per_epoch * 100  \nwarmup_steps     = steps_per_epoch * 7    \ncompleted_steps  = steps_per_epoch * 100   # The offset for resuming at epoch 60\n\n# 2. Safe Custom Learning Rate Schedule with Resume Capability\nclass WarmupCosineDecay(tf.keras.optimizers.schedules.LearningRateSchedule):\n    def __init__(self, base_lr, warmup_steps, total_steps, completed_steps=0, min_lr=1e-6):\n        super().__init__()\n        # Store raw values for safe saving in get_config()\n        self._base_lr = base_lr\n        self._warmup_steps = warmup_steps\n        self._total_steps = total_steps\n        self._completed_steps = completed_steps\n        self._min_lr = min_lr\n\n        # Cast to float32 for calculations\n        self.base_lr_t         = tf.cast(base_lr, tf.float32)\n        self.warmup_steps_t    = tf.cast(warmup_steps, tf.float32)\n        self.total_steps_t     = tf.cast(total_steps, tf.float32)\n        self.completed_steps_t = tf.cast(completed_steps, tf.float32)\n        self.min_lr_t          = tf.cast(min_lr, tf.float32)\n\n    def __call__(self, step):\n        # Shift the internal step forward by the completed steps\n        step = tf.cast(step, tf.float32) + self.completed_steps_t\n        \n        progress = (step - self.warmup_steps_t) / (self.total_steps_t - self.warmup_steps_t)\n        progress = tf.clip_by_value(progress, 0.0, 1.0)\n        \n        cosine_lr = self.min_lr_t + 0.5 * (self.base_lr_t - self.min_lr_t) * (\n            1.0 + tf.cos(3.14159265 * progress)\n        )\n        \n        warmup_lr = self.base_lr_t * (step / self.warmup_steps_t)\n        return tf.where(step < self.warmup_steps_t, warmup_lr, cosine_lr)\n\n    def get_config(self):\n        # Return pure python types, NOT tf tensors\n        return {\n            \"base_lr\":         self._base_lr,\n            \"warmup_steps\":    self._warmup_steps,\n            \"total_steps\":     self._total_steps,\n            \"completed_steps\": self._completed_steps,\n            \"min_lr\":          self._min_lr,\n        }\n\n# 3. Initialize Schedule and Compile\nschedule = WarmupCosineDecay(\n    base_lr=1e-3,\n    warmup_steps=warmup_steps,\n    total_steps=total_steps,\n    completed_steps=completed_steps,   # Pass the offset here\n    min_lr=1e-6\n)\n\nadvanced_optimizer = optimizers.AdamW(\n    learning_rate=schedule,\n    weight_decay=0.01,\n    clipnorm=1.0\n)\n\nmodel.compile(\n    optimizer=advanced_optimizer,\n    loss=losses.CategoricalCrossentropy(\n        from_logits=True,        \n        label_smoothing=0.1\n    ),\n    metrics=[\n        metrics.CategoricalAccuracy(name=\"accuracy\"),\n        metrics.TopKCategoricalAccuracy(k=5, name=\"top5_acc\")\n    ]\n)\n\nmodel.summary()\n\nprint(f\"{model_name} compiled successfully with AdamW + Cosine Warmup Schedule and Label Smoothing.\")\nprint(f\"Steps per epoch : {steps_per_epoch}\")\nprint(f\"Warmup steps    : {warmup_steps}  (7 epochs)\")\nprint(f\"Total steps     : {total_steps} (100 epochs)\")\nprint(f\"Completed steps : {completed_steps} (Starting from epoch 60)\")\nprint(f\"LR range        : 1e-3 → 1e-6 (cosine)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:55.464051Z","iopub.execute_input":"2026-04-26T03:41:55.464443Z","iopub.status.idle":"2026-04-26T03:41:55.564094Z","shell.execute_reply.started":"2026-04-26T03:41:55.464384Z","shell.execute_reply":"2026-04-26T03:41:55.563552Z"}},"outputs":[],"execution_count":null},{"id":"64972773-cff4-46e6-97e2-85c13a0742d6","cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras.utils import plot_model\n\n# ==========================================\n# 1. إنشاء كل المجلدات المطلوبة في مسار المشروع الحالي\n# ==========================================\nos.makedirs(\"Evaluation_Plots\", exist_ok=True)\nos.makedirs(\"logs\", exist_ok=True)\nos.makedirs(\"Saved_Models\", exist_ok=True)\nos.makedirs(\"Wessal_Project\", exist_ok=True)\nos.makedirs(\"training_backup\", exist_ok=True)\n\n# ==========================================\n# 2. حفظ رسمة المعمارية (Architecture Diagram)\n# ==========================================\narchitecture_path = os.path.join(\"Evaluation_Plots\", f\"{model_name}_architecture.png\")\ntry:\n    plot_model(model, to_file=architecture_path, show_shapes=True, show_layer_names=True, expand_nested=True, dpi=300)\n    print(f\"✅ Architecture diagram saved: {architecture_path}\")\nexcept Exception as e:\n    print(\"⚠️ Note: Install 'pydot' and 'graphviz' to generate architecture plots.\")\n\n# ==========================================\n# 3. حفظ بارامترات الموديل\n# ==========================================\nparams = model.count_params()\nparams_path = os.path.join(\"Evaluation_Plots\", f\"{model_name}_params.txt\")\nwith open(params_path, \"w\") as f:\n    f.write(f\"Total Parameters: {params:,}\\n\")\n    f.write(f\"Trainable Parameters: {sum([tf.keras.backend.count_params(w) for w in model.trainable_weights]):,}\\n\")\nprint(f\"✅ Total Parameters: {params:,} (Saved to {params_path})\")\n\n# ==========================================\n# 4. دوال الكول باكس المخصصة (Custom Callbacks)\n# ==========================================\nclass TrainingPlot(callbacks.Callback):\n    def __init__(self, model_name):\n        super().__init__()\n        self.model_name = model_name\n        self.history_dict = {'accuracy': [], 'val_accuracy': [], 'loss': [], 'val_loss': []}\n\n    def on_epoch_end(self, epoch, logs=None):\n        for key in self.history_dict.keys():\n            if key in logs:\n                self.history_dict[key].append(logs[key])\n\n    def on_train_end(self, logs=None):\n        # Accuracy Plot\n        plt.figure(figsize=(10, 5))\n        plt.plot(self.history_dict.get('accuracy', []))\n        plt.plot(self.history_dict.get('val_accuracy', []))\n        plt.title(f\"{self.model_name} - Accuracy Curve\", fontweight='bold')\n        plt.xlabel(\"Epoch\")\n        plt.ylabel(\"Accuracy\")\n        plt.legend([\"Train\", \"Validation\"])\n        plt.grid(True, linestyle='--', alpha=0.7)\n        plt.savefig(os.path.join(\"Evaluation_Plots\", f\"{self.model_name}_accuracy_curve.png\"), dpi=300, bbox_inches='tight')\n        plt.close()\n\n        # Loss Plot\n        plt.figure(figsize=(10, 5))\n        plt.plot(self.history_dict.get('loss', []))\n        plt.plot(self.history_dict.get('val_loss', []))\n        plt.title(f\"{self.model_name} - Loss Curve\", fontweight='bold')\n        plt.xlabel(\"Epoch\")\n        plt.ylabel(\"Loss\")\n        plt.legend([\"Train\", \"Validation\"])\n        plt.grid(True, linestyle='--', alpha=0.7)\n        plt.savefig(os.path.join(\"Evaluation_Plots\", f\"{self.model_name}_loss_curve.png\"), dpi=300, bbox_inches='tight')\n        plt.close()\n\nclass BestMetricLogger(callbacks.Callback):\n    def __init__(self, model_name):\n        super().__init__()\n        self.model_name = model_name\n        self.best_val_acc = 0.0\n\n    def on_epoch_end(self, epoch, logs=None):\n        current_val_acc = logs.get('val_accuracy', 0.0)\n        if current_val_acc > self.best_val_acc:\n            self.best_val_acc = current_val_acc\n\n    def on_train_end(self, logs=None):\n        with open(os.path.join(\"Evaluation_Plots\", f\"{self.model_name}_best_score.txt\"), \"w\") as f:\n            f.write(f\"Best Validation Accuracy: {self.best_val_acc:.4f}\\n\")\n\n# ==========================================\n# 5. قائمة الكول باكس النهائية (Final Callbacks List)\n# ==========================================\nmodel_callbacks = [\n    callbacks.ModelCheckpoint(\n        # تم تعديل الامتداد إلى .keras والمسار ليكون صحيحاً\n        filepath=os.path.join(\"Saved_Models\", f\"{model_name}_best.keras\"),\n        monitor='val_accuracy',\n        save_best_only=True,\n        mode='max',\n        verbose=1\n    ),\n    callbacks.EarlyStopping(\n        monitor='val_accuracy',\n        patience=20,    \n        restore_best_weights=True,\n        verbose=1\n    ),\n    callbacks.CSVLogger(\n        filename=os.path.join(\"Wessal_Project\", f\"{model_name}_log.csv\"), \n        append=False\n    ),\n    TrainingPlot(model_name),\n    BestMetricLogger(model_name),\n    tf.keras.callbacks.BackupAndRestore(backup_dir='./training_backup'),\n    callbacks.TensorBoard(log_dir=os.path.join(\"logs\", model_name), histogram_freq=1)\n]\n\nprint(\"✅ Callbacks and directories are fully configured and ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:41:55.565033Z","iopub.execute_input":"2026-04-26T03:41:55.565368Z","iopub.status.idle":"2026-04-26T03:42:04.979932Z","shell.execute_reply.started":"2026-04-26T03:41:55.565345Z","shell.execute_reply":"2026-04-26T03:42:04.979225Z"}},"outputs":[],"execution_count":null},{"id":"0f6f083e","cell_type":"code","source":"# Define the maximum number of epochs\n# (EarlyStopping will likely stop it much earlier, usually around 30-50 epochs)\nTRAINING_EPOCHS = 200\n\nprint(f\"Maximum Epochs set to: {TRAINING_EPOCHS}\")\nprint(f\"Batch Size (handled by tf.data): {BATCH_SIZE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:42:04.980994Z","iopub.execute_input":"2026-04-26T03:42:04.981360Z","iopub.status.idle":"2026-04-26T03:42:04.985827Z","shell.execute_reply.started":"2026-04-26T03:42:04.981334Z","shell.execute_reply":"2026-04-26T03:42:04.985085Z"}},"outputs":[],"execution_count":null},{"id":"efbba74d-ccff-4260-89e1-33319006f98b","cell_type":"code","source":"print(f\"Training for: {model_name}...\")\n\n# Start the training process\nhistory = model.fit(\n    train_dataset,\n    validation_data=val_dataset,\n    initial_epoch=100,\n    epochs=TRAINING_EPOCHS,\n    callbacks=model_callbacks,\n    verbose=1\n)\n\nprint(f\"\\n Training Phase Completed for {model_name}!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T03:42:04.986901Z","iopub.execute_input":"2026-04-26T03:42:04.987247Z"}},"outputs":[],"execution_count":null},{"id":"fd59f26b","cell_type":"markdown","source":"### 1. Training History\n| Goal | Outputs |\n| :--- | :--- |\n| Trace the training process and detect overfitting/underfitting. | Accuracy and Loss curves over epochs. |","metadata":{}},{"id":"0b829b7d","cell_type":"code","source":"print(f\"--- Visualizing Training History for {model_name} ---\")\n\n# Load training history from the saved CSV log\nlog_path = os.path.join(\"../Training_Histories\", f\"{model_name}_log.csv\")\n\nif os.path.exists(log_path):\n    history_df = pd.read_csv(log_path)\n    \n    fig, axes = plt.subplots(1, 2, figsize=(16, 5))\n    \n    # Accuracy Plot\n    axes[0].plot(history_df['epoch'], history_df['accuracy'], label='Train Accuracy', color='#4285F4', linewidth=2)\n    axes[0].plot(history_df['epoch'], history_df['val_accuracy'], label='Validation Accuracy', color='#34A853', linewidth=2)\n    axes[0].set_title('Model Accuracy over Epochs', fontweight='bold')\n    axes[0].set_xlabel('Epoch')\n    axes[0].set_ylabel('Accuracy')\n    axes[0].legend()\n    axes[0].grid(True, linestyle='--', alpha=0.6)\n    \n    # Loss Plot\n    axes[1].plot(history_df['epoch'], history_df['loss'], label='Train Loss', color='#EA4335', linewidth=2)\n    axes[1].plot(history_df['epoch'], history_df['val_loss'], label='Validation Loss', color='#FBBC05', linewidth=2)\n    axes[1].set_title('Model Loss over Epochs', fontweight='bold')\n    axes[1].set_xlabel('Epoch')\n    axes[1].set_ylabel('Loss')\n    axes[1].legend()\n    axes[1].grid(True, linestyle='--', alpha=0.6)\n    \n    plt.tight_layout()\n    plt.savefig(os.path.join(\"../Evaluation_Plots\", f\"{model_name}_Training_History.png\"), dpi=300)\n    plt.show()\nelse:\n    print(f\"Log file not found at {log_path}. Ensure the model has been trained.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"280f3cbf-409d-401d-9443-275fe730ec14","cell_type":"code","source":"# حفظ الموديل بعد انتهاء التدريب\nmodel.save('my_model.keras')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"f595f21c-e04b-42e7-8762-1e6694fbc306","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}