{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":5315518,"sourceType":"datasetVersion","datasetId":3036481},{"sourceId":14133088,"sourceType":"datasetVersion","datasetId":9005742},{"sourceId":619157,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":465652,"modelId":481483}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q mediapipe opencv-python gradio gtts transformers torch  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:43:27.561026Z","iopub.execute_input":"2026-02-10T13:43:27.562000Z","iopub.status.idle":"2026-02-10T13:43:36.222169Z","shell.execute_reply.started":"2026-02-10T13:43:27.561959Z","shell.execute_reply":"2026-02-10T13:43:36.220997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Wipe out the potentially corrupted versions\n!pip uninstall -q -y mediapipe protobuf\n\n# 2. Install specific versions known to work with MediaPipe 0.10.x\n!pip install -q mediapipe==0.10.11 protobuf==3.20.3\n\n# 3. Finalize with a compatible NumPy (to prevent that earlier error from returning)\n!pip install -q \"numpy<2.0\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:43:36.223928Z","iopub.execute_input":"2026-02-10T13:43:36.225017Z","iopub.status.idle":"2026-02-10T13:43:53.533742Z","shell.execute_reply.started":"2026-02-10T13:43:36.224976Z","shell.execute_reply":"2026-02-10T13:43:53.532456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport mediapipe as mp\nimport cv2\nimport os\nimport json\nfrom gtts import gTTS\nimport gradio as gr\nfrom transformers import pipeline\n\n# Suppress Warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:43:53.535006Z","iopub.execute_input":"2026-02-10T13:43:53.535499Z","iopub.status.idle":"2026-02-10T13:44:18.213326Z","shell.execute_reply.started":"2026-02-10T13:43:53.535461Z","shell.execute_reply":"2026-02-10T13:44:18.212188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"INPUT_SIZE = 64\nN_ROWS = 543\nN_DIMS = 3\nN_COLS = 0 # Will be updated after IDXS definition\nNUM_CLASSES = 250\nROWS_PER_FRAME = 543\n\n# Model Hyperparameters\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\nUNITS = 512\nNUM_BLOCKS = 2\nMLP_RATIO = 2\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initializers and Activations\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\nGELU = tf.keras.activations.gelu","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:44:18.215197Z","iopub.execute_input":"2026-02-10T13:44:18.216074Z","iopub.status.idle":"2026-02-10T13:44:18.222838Z","shell.execute_reply.started":"2026-02-10T13:44:18.216042Z","shell.execute_reply":"2026-02-10T13:44:18.221534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"USE_TYPES = ['left_hand', 'pose', 'right_hand']\nSTART_IDX = 468\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n# Landmark indices in original data\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n\n# Landmark indices in processed data\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\nLIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:44:18.224089Z","iopub.execute_input":"2026-02-10T13:44:18.224369Z","iopub.status.idle":"2026-02-10T13:44:18.251299Z","shell.execute_reply.started":"2026-02-10T13:44:18.224345Z","shell.execute_reply":"2026-02-10T13:44:18.250138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/asl-signs/train.csv')\ntrain_df['sign_ord'] = train_df['sign'].astype('category').cat.codes\nORD2SIGN = train_df[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()\n\ndel train_df\n\nSTATS_PATH = '/kaggle/input/datasets/idowuadamo/isolatedasl-data-stats'\n\nLIPS_MEAN = np.load(f'{STATS_PATH}/LIPS_MEAN.npy')\nLIPS_STD = np.load(f'{STATS_PATH}/LIPS_STD.npy')\nLEFT_HANDS_MEAN = np.load(f'{STATS_PATH}/LEFT_HANDS_MEAN.npy')\nLEFT_HANDS_STD = np.load(f'{STATS_PATH}/LEFT_HANDS_STD.npy')\nPOSE_MEAN = np.load(f'{STATS_PATH}/POSE_MEAN.npy')\nPOSE_STD = np.load(f'{STATS_PATH}/POSE_STD.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:46:22.255543Z","iopub.execute_input":"2026-02-10T13:46:22.256418Z","iopub.status.idle":"2026-02-10T13:46:22.496476Z","shell.execute_reply.started":"2026-02-10T13:46:22.256383Z","shell.execute_reply":"2026-02-10T13:46:22.495448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        normalisation_correction = tf.constant([\n                    # Add 0.50 to left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                    [0] * len(LIPS_IDXS) + [0.50] * len(LEFT_HAND_IDXS) + [0.50] * len(POSE_IDXS),\n                    # Y coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                    # Z coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                ],\n                dtype=tf.float32,\n            )\n        self.normalisation_correction = tf.transpose(normalisation_correction, [1,0])\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_ROWS,N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand by comparing summed absolute coordinates\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non-NaN Hand values in each frame for the dominant hand\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        \n        # Find frames indices with coordinates of the dominant hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter frames\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LANDMARK_IDXS_LEFT_DOMINANT0, axis=1)\n        else:\n            data = tf.gather(data, LANDMARK_IDXS_RIGHT_DOMINANT0, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n        \n        # Video fits in INPUT_SIZE\n        if N_FRAMES < INPUT_SIZE:\n            # Pad With -1 to indicate padding\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            # Pad Data With Zeros\n            data = tf.pad(data, [[0, INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        # Video needs to be downsampled to INPUT_SIZE\n        else:\n            # Repeat\n            if N_FRAMES < INPUT_SIZE**2:\n                repeats = tf.math.floordiv(INPUT_SIZE * INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), INPUT_SIZE)\n            if tf.math.mod(len(data), INPUT_SIZE) > 0:\n                pool_size += 1\n\n            if pool_size == 1:\n                pad_size = (pool_size * INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [INPUT_SIZE, -1, N_COLS, N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:46:29.904962Z","iopub.execute_input":"2026-02-10T13:46:29.905340Z","iopub.status.idle":"2026-02-10T13:46:29.967167Z","shell.execute_reply.started":"2026-02-10T13:46:29.905311Z","shell.execute_reply":"2026-02-10T13:46:29.966082Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Classes (Transformers and Embeddings)","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n        \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:46:34.619466Z","iopub.execute_input":"2026-02-10T13:46:34.620335Z","iopub.status.idle":"2026-02-10T13:46:34.624912Z","shell.execute_reply.started":"2026-02-10T13:46:34.620301Z","shell.execute_reply":"2026-02-10T13:46:34.623897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    # calculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention\n\nclass Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # Multi-Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for mha, mlp in zip(self.mhas, self.mlps):\n            x = x + mha(x, attention_mask)\n            x = x + mlp(x)\n    \n        return x\n\nclass LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initialized with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )\n\nclass Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([3], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((\n            lips_embedding, left_hand_embedding, pose_embedding,\n        ), axis=3)\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        max_frame_idxs = tf.clip_by_value(\n                tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True),\n                1,\n                np.PINF,\n            )\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / max_frame_idxs * INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:46:36.959494Z","iopub.execute_input":"2026-02-10T13:46:36.960294Z","iopub.status.idle":"2026-02-10T13:46:36.982272Z","shell.execute_reply.started":"2026-02-10T13:46:36.960259Z","shell.execute_reply":"2026-02-10T13:46:36.981341Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Loading","metadata":{}},{"cell_type":"code","source":"def scce_with_ls(y_true, y_pred):\n    y_true = tf.cast(y_true, tf.int32)\n    y_true = tf.one_hot(y_true, NUM_CLASSES, axis=1) \n    return tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=0.25) \n\ndef get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n\n    # Padding Mask - WRAPPED IN LAMBDA LAYERS\n    mask0 = tf.keras.layers.Lambda(lambda x: tf.cast(tf.math.not_equal(x, -1), tf.float32), name='mask0')(non_empty_frame_idxs)\n    mask0_expanded = tf.keras.layers.Lambda(lambda x: tf.expand_dims(x, axis=2), name='mask0_expanded')(mask0)\n\n    def create_mask(x):\n        # x[0] is mask0_expanded, x[1] is mask0\n        mask = tf.where(\n            (tf.random.uniform(tf.shape(x[0])) > 0.25) & tf.math.not_equal(x[0], 0.0),\n            1.0,\n            0.0,\n        )\n        # Correct Samples Which are all masked now...\n        mask = tf.where(\n            tf.math.equal(tf.reduce_sum(mask, axis=[1,2], keepdims=True), 0.0),\n            x[0], # use mask0_expanded\n            mask,\n        )\n        return mask\n    \n    mask = tf.keras.layers.Lambda(create_mask, name='create_mask')([mask0_expanded, mask0_expanded]) \n\n    # Slicing the XY coordinates\n    x = tf.keras.layers.Lambda(lambda t: tf.slice(t, [0,0,0,0], [-1,INPUT_SIZE, N_COLS, 2]), name='slice_xy')(frames)\n    \n    # LIPS \n    lips = tf.keras.layers.Lambda(lambda t: tf.slice(t, [0,0,LIPS_START,0], [-1,INPUT_SIZE, 40, 2]), name='slice_lips')(x)\n    lips = tf.keras.layers.Lambda(\n        lambda t: tf.where(\n            tf.math.equal(t, 0.0), 0.0, (t - LIPS_MEAN) / LIPS_STD\n        ), name='normalize_lips')(lips)\n    lips = tf.keras.layers.Reshape((INPUT_SIZE, 40*2), name='reshape_lips')(lips)\n\n    # LEFT HAND\n    left_hand = tf.keras.layers.Lambda(lambda t: tf.slice(t, [0,0,40,0], [-1,INPUT_SIZE, 21, 2]), name='slice_left_hand')(x)\n    left_hand = tf.keras.layers.Lambda(\n        lambda t: tf.where(\n            tf.math.equal(t, 0.0), 0.0, (t - LEFT_HANDS_MEAN) / LEFT_HANDS_STD\n        ), name='normalize_left_hand')(left_hand)\n    left_hand = tf.keras.layers.Reshape((INPUT_SIZE, 21*2), name='reshape_left_hand')(left_hand)\n\n    # POSE\n    pose = tf.keras.layers.Lambda(lambda t: tf.slice(t, [0,0,61,0], [-1,INPUT_SIZE, 5, 2]), name='slice_pose')(x)\n    pose = tf.keras.layers.Lambda(\n        lambda t: tf.where(\n            tf.math.equal(t, 0.0), 0.0, (t - POSE_MEAN) / POSE_STD\n        ), name='normalize_pose')(pose)\n    pose = tf.keras.layers.Reshape((INPUT_SIZE, 5*2), name='reshape_pose')(pose)\n    \n    # Embedding\n    x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n    \n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    \n    # Pooling\n    def masked_pooling(tensors):\n        x_tensor, mask_tensor = tensors\n        return tf.reduce_sum(x_tensor * mask_tensor, axis=1) / tf.reduce_sum(mask_tensor, axis=1)\n        \n    x = tf.keras.layers.Lambda(masked_pooling, name='masked_pooling')([x, mask])\n    \n    # Classifier Dropout\n    x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n    # Classification Layer\n    x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax', kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    \n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Compile (Optional for inference, but keeps loading consistent)\n    loss = scce_with_ls\n    optimizer = tf.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    metrics = [tf.keras.metrics.SparseCategoricalAccuracy(name='acc')]\n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model\n\n# Initialize and Load\nmodel = get_model()\n\n# UPDATE PATH TO YOUR SAVED WEIGHTS\nmodel_weights_path = '/kaggle/input/original-model/tensorflow2/default/1/model.weights.h5' \nif os.path.exists(model_weights_path):\n    model.load_weights(model_weights_path)\n    print(\"Model weights loaded successfully.\")\nelse:\n    print(f\"Error: Model weights not found at {model_weights_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:46:42.334532Z","iopub.execute_input":"2026-02-10T13:46:42.334868Z","iopub.status.idle":"2026-02-10T13:46:46.102331Z","shell.execute_reply.started":"2026-02-10T13:46:42.334846Z","shell.execute_reply":"2026-02-10T13:46:46.101288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Mediapipe and Frame Extraction","metadata":{}},{"cell_type":"code","source":"# Setup for MediaPipe Holistic\nmp_holistic = mp.solutions.holistic\nholistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5) \n\n# Extract Landmarks From Frame\ndef extract_landmarks(image):\n    image_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    results = holistic.process(image_rgb)\n    landmarks = []\n    # Face\n    if results.face_landmarks:\n        for lm in results.face_landmarks.landmark:\n            landmarks.append([lm.x, lm.y, lm.z if hasattr(lm, 'z') else 0])\n    else:\n        landmarks.extend([[0,0,0]] * 468)\n    # Left hand\n    if results.left_hand_landmarks:\n        for lm in results.left_hand_landmarks.landmark:\n            landmarks.append([lm.x, lm.y, lm.z if hasattr(lm, 'z') else 0])\n    else:\n        landmarks.extend([[0,0,0]] * 21)\n    # Pose\n    if results.pose_landmarks:\n        for lm in results.pose_landmarks.landmark:\n            landmarks.append([lm.x, lm.y, lm.z if hasattr(lm, 'z') else 0])\n    else:\n        landmarks.extend([[0,0,0]] * 33)\n    # Right hand\n    if results.right_hand_landmarks:\n        for lm in results.right_hand_landmarks.landmark:\n            landmarks.append([lm.x, lm.y, lm.z if hasattr(lm, 'z') else 0])\n    else:\n        landmarks.extend([[0,0,0]] * 21)\n    landmarks = landmarks[:543]\n    return np.array(landmarks, dtype=np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:46:46.104086Z","iopub.execute_input":"2026-02-10T13:46:46.104503Z","iopub.status.idle":"2026-02-10T13:46:46.228332Z","shell.execute_reply.started":"2026-02-10T13:46:46.104472Z","shell.execute_reply":"2026-02-10T13:46:46.222963Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LLM Setup","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nfrom huggingface_hub import login\n\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"HF_TOKEN\")\nlogin(token=secret_value_0, new_session=False) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:49:25.199816Z","iopub.execute_input":"2026-02-10T13:49:25.201791Z","iopub.status.idle":"2026-02-10T13:49:25.504106Z","shell.execute_reply.started":"2026-02-10T13:49:25.201718Z","shell.execute_reply":"2026-02-10T13:49:25.502939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Text Generation Model Setup\nllm = pipeline(\"text-generation\", model=\"Qwen/Qwen2.5-3B-Instruct\", device_map=\"auto\")\n\ndef refine_to_sentence_fewshot(sign_sequence):\n    \"\"\"\n    Convert ASL sign sequence to English using multi-shot prompting with strict output formatting.\n    \"\"\"\n    if not sign_sequence:\n        return \"Signs Detected: None\\nSentence: No signs detected.\"\n    \n    llm_input = ' '.join(sign_sequence)\n    \n    prompt = (\n        \"You are a strictly formatted ASL-to-English translation engine. \"\n        \"Your task is to convert American Sign Language (ASL) glosses into a coherent English sentence.\\n\\n\"\n        \n        \"RULES:\\n\"\n        \"1. Output exactly two lines: 'Signs Detected:' followed by the input, and 'Sentence:' followed by the translation.\\n\"\n        \"2. Do not add introductions, explanations, or extra text.\\n\"\n        \"3. If the input is random noise, incoherent words, or insufficient to form a thought, output '[Unintelligible]'.\\n\\n\"\n        \n        \"EXAMPLES:\\n\"\n        \"Input: you name what you\\n\"\n        \"Signs Detected: you name what you\\n\"\n        \"Sentence: What is your name?\\n\\n\"\n        \n        \"Input: school go morning i\\n\"\n        \"Signs Detected: school go morning i\\n\"\n        \"Sentence: I go to school in the morning.\\n\\n\"\n        \n        \"Input: purple monkey dishwasher\\n\"\n        \"Signs Detected: purple monkey dishwasher\\n\"\n        \"Sentence: [Unintelligible]\\n\\n\"\n        \n        \"Input: please help me\\n\"\n        \"Signs Detected: please help me\\n\"\n        \"Sentence: Please help me.\\n\\n\"\n        \n        f\"Input: {llm_input}\\n\"\n    )\n    \n    try:\n        resp = llm(prompt, max_new_tokens=100, do_sample=False)[0]['generated_text']\n        \n        # PARSING LOGIC:\n        if \"Input: \" + llm_input in resp:\n            output_section = resp.split(f\"Input: {llm_input}\")[-1].strip()\n        else:\n            output_section = resp.strip()\n\n        # Clean up any trailing examples\n        final_output = output_section.split(\"Input:\")[0].strip()\n        return final_output\n\n    except Exception as e:\n        return f\"Signs Detected: {llm_input}\\nSentence: Error processing translation.\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:50:12.297390Z","iopub.execute_input":"2026-02-10T13:50:12.298646Z","iopub.status.idle":"2026-02-10T13:51:05.654486Z","shell.execute_reply.started":"2026-02-10T13:50:12.298582Z","shell.execute_reply":"2026-02-10T13:51:05.653257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Text to Speech\ndef text_to_audio(text, audio_path='output.mp3'):\n    # Extract just the sentence part for TTS if it's in the structured format\n    if \"Sentence:\" in text:\n        text_to_speak = text.split(\"Sentence:\")[-1].strip()\n    else:\n        text_to_speak = text\n        \n    if text_to_speak == \"[Unintelligible]\":\n        return None\n        \n    tts = gTTS(text_to_speak, lang='en')\n    tts.save(audio_path)\n    return audio_path \n\n# Video and Live Frame Processing\ndef process_uploaded_video(video_path):\n    if video_path is None:\n        return \"No video uploaded.\", None\n    cap = cv2.VideoCapture(video_path)\n    frames_landmarks = []\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n        landmarks = extract_landmarks(frame)\n        frames_landmarks.append(landmarks)\n    cap.release()\n    \n    if not frames_landmarks:\n        return \"No frames detected.\", None\n    \n    data = np.array(frames_landmarks)\n    window_size = INPUT_SIZE\n    step = 16\n    predictions = []\n    \n    # Process video in chunks\n    for start in range(0, max(1, len(data) - window_size + 1), step):\n        window_data = data[start:start+window_size]\n        # Pad last chunk if needed\n        if len(window_data) < window_size:\n            pad = np.zeros((window_size - len(window_data), 543, 3), dtype=np.float32)\n            window_data = np.concatenate((window_data, pad), axis=0)\n            \n        processed_data, non_empty_idxs = preprocess_layer(window_data)\n        processed_data = np.expand_dims(processed_data, 0)\n        non_empty_idxs = np.expand_dims(non_empty_idxs, 0)\n        \n        pred = model.predict({'frames': processed_data, 'non_empty_frame_idxs': non_empty_idxs}, verbose=0)[0].argmax()\n        sign = ORD2SIGN.get(pred, 'unknown')\n        predictions.append(sign)\n        \n    # Remove consecutive duplicates and 'unknown'\n    unique_seq = []\n    for s in predictions:\n        if not unique_seq or (s != unique_seq[-1] and s != 'unknown'):\n            unique_seq.append(s)\n            \n    sentence = refine_to_sentence_fewshot(unique_seq)\n    audio_path = text_to_audio(sentence)\n    return sentence, audio_path\n\n# Globals for Live Stream\nframe_buffer = []\nsign_sequence = []\n\ndef process_live_frame(image):\n    global frame_buffer, sign_sequence\n    if image is None:\n        return \"No frame.\", None\n        \n    landmarks = extract_landmarks(image)\n    frame_buffer.append(landmarks)\n    \n    # Wait until buffer fills\n    if len(frame_buffer) >= INPUT_SIZE:\n        data = np.array(frame_buffer[-INPUT_SIZE:])\n        processed_data, non_empty_idxs = preprocess_layer(data)\n        processed_data = np.expand_dims(processed_data, 0)\n        non_empty_idxs = np.expand_dims(non_empty_idxs, 0)\n        \n        pred = model.predict({'frames': processed_data, 'non_empty_frame_idxs': non_empty_idxs}, verbose=0)[0].argmax()\n        sign = ORD2SIGN.get(pred, 'unknown')\n        \n        if sign != 'unknown' and (not sign_sequence or sign != sign_sequence[-1]):\n            sign_sequence.append(sign)\n        \n        # Sliding window: keep last 48 frames (overlap)\n        frame_buffer[:] = frame_buffer[16:] \n        \n    # Trigger Translation every ~10 signs or manually? \n    # For now, we return the current sequence.\n    current_status = ' '.join(sign_sequence) if sign_sequence else \"Detecting...\"\n    \n    # Auto-translate if sequence gets long (optional logic)\n    if len(sign_sequence) > 0 and len(sign_sequence) % 10 == 0:\n         # Note: In a real app, you might want a specific 'End' gesture to trigger this\n         pass\n         \n    return current_status, None\n\ndef trigger_translation():\n    global sign_sequence\n    sentence = refine_to_sentence_fewshot(sign_sequence)\n    audio_path = text_to_audio(sentence)\n    # Optional: Clear sequence after translation\n    # sign_sequence = [] \n    return sentence, audio_path\n\ndef clear_buffer():\n    global frame_buffer, sign_sequence\n    frame_buffer = []\n    sign_sequence = []\n    return \"Cleared.\", None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:51:05.656839Z","iopub.execute_input":"2026-02-10T13:51:05.657246Z","iopub.status.idle":"2026-02-10T13:51:05.674198Z","shell.execute_reply.started":"2026-02-10T13:51:05.657196Z","shell.execute_reply":"2026-02-10T13:51:05.673025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with gr.Blocks() as demo:\n    gr.Markdown(\"# Sign Language to Text and Audio\")\n    with gr.Tabs():\n        # TAB 1: Video Upload\n        with gr.Tab(\"Upload Video\"):\n            video_input = gr.Video(label=\"Upload Sign Language Video\")\n            upload_output_text = gr.Text(label=\"Translated Sentence\")\n            upload_output_audio = gr.Audio(label=\"Audio Output\")\n            video_input.change(\n                fn=process_uploaded_video,\n                inputs=video_input,\n                outputs=[upload_output_text, upload_output_audio]\n            )\n            \n        # TAB 2: Webcam\n        with gr.Tab(\"Live Webcam\"):\n            with gr.Row():\n                webcam_input = gr.Image(sources=[\"webcam\"], streaming=True)\n                with gr.Column():\n                    live_status_text = gr.Text(label=\"Detected Glosses (Live)\")\n                    live_translated_text = gr.Text(label=\"Final Translation\")\n                    live_output_audio = gr.Audio(label=\"Speech\")\n            \n            with gr.Row():\n                translate_btn = gr.Button(\"Translate Now\", variant=\"primary\")\n                clear_btn = gr.Button(\"Clear Buffer\")\n\n            # Stream frames\n            webcam_input.stream(\n                fn=process_live_frame,\n                inputs=webcam_input,\n                outputs=[live_status_text, live_output_audio],\n                stream_every=0.1,\n                time_limit=60\n            )\n            \n            # Buttons\n            translate_btn.click(\n                fn=trigger_translation,\n                inputs=None,\n                outputs=[live_translated_text, live_output_audio]\n            )\n            \n            clear_btn.click(\n                fn=clear_buffer,\n                inputs=None,\n                outputs=[live_status_text, live_output_audio]\n            )\n\ndemo.launch()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T13:51:11.520944Z","iopub.execute_input":"2026-02-10T13:51:11.521309Z","iopub.status.idle":"2026-02-10T13:51:12.403548Z","shell.execute_reply.started":"2026-02-10T13:51:11.521279Z","shell.execute_reply":"2026-02-10T13:51:12.402429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}