{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314},{"sourceType":"datasetVersion","sourceId":5315518,"datasetId":3036481,"databundleVersionId":5388737},{"sourceType":"modelInstanceVersion","sourceId":832816,"databundleVersionId":16692058,"modelInstanceId":633504,"modelId":645428}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q mediapipe gradio gtts jiwer transformers accelerate bitsandbytes","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q mediapipe==0.10.14","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import mediapipe as mp\n\nmp_holistic = mp.solutions.holistic ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:33:23.685943Z","iopub.execute_input":"2026-04-19T08:33:23.686324Z","iopub.status.idle":"2026-04-19T08:33:58.779164Z","shell.execute_reply.started":"2026-04-19T08:33:23.686289Z","shell.execute_reply":"2026-04-19T08:33:58.77847Z"}},"outputs":[{"name":"stderr","text":"2026-04-19 08:33:28.233239: E external/local_xla/xla/stream_executor/cuda/cuda_fft.cc:467] Unable to register cuFFT factory: Attempting to register factory for plugin cuFFT when one has already been registered\nWARNING: All log messages before absl::InitializeLog() is called are written to STDERR\nE0000 00:00:1776587608.710804     117 cuda_dnn.cc:8579] Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered\nE0000 00:00:1776587608.820172     117 cuda_blas.cc:1407] Unable to register cuBLAS factory: Attempting to register factory for plugin cuBLAS when one has already been registered\nW0000 00:00:1776587609.966214     117 computation_placer.cc:177] computation placer already registered. Please check linkage and avoid linking the same target more than once.\nW0000 00:00:1776587609.966253     117 computation_placer.cc:177] computation placer already registered. Please check linkage and avoid linking the same target more than once.\nW0000 00:00:1776587609.966257     117 computation_placer.cc:177] computation placer already registered. Please check linkage and avoid linking the same target more than once.\nW0000 00:00:1776587609.966259     117 computation_placer.cc:177] computation placer already registered. Please check linkage and avoid linking the same target more than once.\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"import tensorflow as tf\n\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n        print(\"TensorFlow memory growth enabled.\")\n    except RuntimeError as e:\n        print(f\"TF Memory Error: {e}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:34:59.070063Z","iopub.execute_input":"2026-04-19T08:34:59.071286Z","iopub.status.idle":"2026-04-19T08:35:01.248956Z","shell.execute_reply.started":"2026-04-19T08:34:59.071251Z","shell.execute_reply":"2026-04-19T08:35:01.247882Z"}},"outputs":[{"name":"stdout","text":"TensorFlow memory growth enabled.\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"import os\nimport time\nimport json\nimport numpy as np\nimport tensorflow as tf\nimport mediapipe as mp\nimport cv2\nimport gradio as gr\nfrom gtts import gTTS\nfrom transformers import pipeline\nfrom huggingface_hub import login\n\nprint(f'Tensorflow V{tf.__version__}')\n\n# Configurations\nINPUT_SIZE = 128 \nN_DIMS = 3\nNUM_CLASSES = 250\nMODEL_WEIGHTS_PATH = '/kaggle/input/models/idowuadamo/hybrid-model-toptier/tensorflow2/default/1/hybrid_model_abs.weights.h5'\n\n# Landmark indices\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\n\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\nN_COLS_FINAL = N_COLS * 9 \n\ntry:\n    json_path = \"/kaggle/input/competitions/asl-signs/sign_to_prediction_index_map.json\" #\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\"\n    with open(json_path, 'r') as f:\n        data = json.load(f)\n        ORD2SIGN = {v: k for k, v in data.items()}\n        print(\"Label map loaded successfully.\")\nexcept Exception as e:\n    print(f\"Could not load map from {json_path}. Error: {e}\")\n    ORD2SIGN = {i: f\"sign_{i}\" for i in range(NUM_CLASSES)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:35:12.655378Z","iopub.execute_input":"2026-04-19T08:35:12.655703Z","iopub.status.idle":"2026-04-19T08:35:42.658546Z","shell.execute_reply.started":"2026-04-19T08:35:12.655676Z","shell.execute_reply":"2026-04-19T08:35:42.657761Z"}},"outputs":[{"name":"stdout","text":"Tensorflow V2.19.0\nLabel map loaded successfully.\n","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"class PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__() \n        self.lips_idxs = LIPS_IDXS\n        self.left_hand_idxs = LEFT_HAND_IDXS\n        self.pose_idxs = POSE_IDXS\n        self.landmark_idxs_left = LANDMARK_IDXS_LEFT_DOMINANT0\n        self.landmark_idxs_right = LANDMARK_IDXS_RIGHT_DOMINANT0\n        \n    @tf.function\n    def call(self, data0):\n        data0 = tf.cast(data0, tf.float32)\n\n        # Filter dominant hand \n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0.0, 1.0))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0.0, 1.0))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0.0, 1.0), axis=[1, 2]\n            )\n            data = tf.gather(data0, self.landmark_idxs_left, axis=1)\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0.0, 1.0), axis=[1, 2]\n            )\n            data = tf.gather(data0, self.landmark_idxs_right, axis=1)\n            # Mirror X coordinate for right-handers\n            data = tf.concat([\n                -1.0 * tf.expand_dims(data[:, :, 0], axis=-1),\n                tf.expand_dims(data[:, :, 1], axis=-1),\n                tf.expand_dims(data[:, :, 2], axis=-1)\n            ], axis=-1)\n            \n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        data = tf.gather(data, non_empty_frames_idxs, axis=0)\n\n        # Normalization (Anchor to Lips/Nose)\n        lips = data[:, :40, :] \n        lips_mean = tf.math.reduce_mean(tf.where(tf.math.is_nan(lips), 0.0, lips), axis=1, keepdims=True)\n        lips_std = tf.math.reduce_std(tf.where(tf.math.is_nan(data), 0.0, data), axis=[1,2], keepdims=True) + 1e-6\n        data = (data - lips_mean) / lips_std\n        data = tf.where(tf.math.is_nan(data), 0.0, data)\n\n        # Resizing / Interpolation\n        N_FRAMES = tf.shape(data)[0]\n        if N_FRAMES < INPUT_SIZE:\n            non_empty_frames_idxs = tf.pad(\n                tf.cast(non_empty_frames_idxs, tf.float32), \n                [[0, INPUT_SIZE - N_FRAMES]], constant_values=-1\n            )\n            data = tf.pad(data, [[0, INPUT_SIZE - N_FRAMES], [0,0], [0,0]], constant_values=0)\n        else:\n            data_flat = tf.reshape(data, [1, N_FRAMES, -1, 1])\n            data_resized = tf.image.resize(\n                data_flat, [INPUT_SIZE, tf.shape(data_flat)[2]], method=tf.image.ResizeMethod.BILINEAR\n            )\n            data = tf.reshape(data_resized, [INPUT_SIZE, -1, N_DIMS])\n            non_empty_frames_idxs = tf.linspace(0.0, tf.cast(N_FRAMES, tf.float32), INPUT_SIZE)\n\n        # Motion Features\n        dx = data[1:, :, :] - data[:-1, :, :]\n        dx = tf.concat([tf.zeros_like(data[:1, :, :]), dx], axis=0)\n        \n        ddx = dx[1:, :, :] - dx[:-1, :, :]\n        ddx = tf.concat([tf.zeros_like(dx[:1, :, :]), ddx], axis=0)\n        \n        data = tf.concat([data, dx, ddx], axis=-1)\n        data = tf.reshape(data, (INPUT_SIZE, -1))\n        \n        return data, non_empty_frames_idxs\n\npreprocess_layer = PreprocessLayer()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:35:55.457791Z","iopub.execute_input":"2026-04-19T08:35:55.45841Z","iopub.status.idle":"2026-04-19T08:35:55.473863Z","shell.execute_reply.started":"2026-04-19T08:35:55.458364Z","shell.execute_reply":"2026-04-19T08:35:55.473189Z"}},"outputs":[],"execution_count":5},{"cell_type":"code","source":"# Custom Layers\nclass EcaLayer(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=5, **kwargs):\n        super().__init__(**kwargs)\n        self.conv = tf.keras.layers.Conv1D(1, kernel_size=kernel_size, padding='same', use_bias=False)\n\n    def call(self, x):\n        attn = tf.reduce_mean(x, axis=1, keepdims=True)\n        attn = tf.transpose(attn, (0, 2, 1))\n        attn = self.conv(attn)\n        attn = tf.transpose(attn, (0, 2, 1))\n        return x * tf.math.sigmoid(attn)\n\nclass Conv1DBlock(tf.keras.layers.Layer):\n    def __init__(self, dim, kernel_size=11, drop_rate=0.2, expand=4):\n        super().__init__()\n        self.conv = tf.keras.layers.DepthwiseConv1D(kernel_size, padding='same', use_bias=False)\n        self.bn = tf.keras.layers.BatchNormalization()\n        self.act = tf.keras.layers.Activation('swish')\n        self.se = EcaLayer(kernel_size=5)\n        self.project = tf.keras.layers.Dense(dim, use_bias=False)\n        self.drop = tf.keras.layers.Dropout(drop_rate)\n        \n    def call(self, x, training=None):\n        skip = x\n        x = self.conv(x)\n        x = self.bn(x, training=training)\n        x = self.act(x)\n        x = self.se(x)\n        x = self.project(x)\n        if training: x = self.drop(x)\n        return x + skip\n\nclass TransformerBlock(tf.keras.layers.Layer):\n    def __init__(self, embed_dim, num_heads, ff_dim, rate=0.1):\n        super().__init__()\n        self.att = tf.keras.layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = tf.keras.Sequential([\n            tf.keras.layers.Dense(ff_dim, activation=\"gelu\"),\n            tf.keras.layers.Dense(embed_dim),\n        ])\n        self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = tf.keras.layers.Dropout(rate)\n        self.dropout2 = tf.keras.layers.Dropout(rate)\n\n    def call(self, inputs, training=None):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output) \n\nclass LearnablePositionalEmbedding(tf.keras.layers.Layer):\n    def __init__(self, max_len, embed_dim):\n        super().__init__()\n        self.pos_embedding = tf.keras.layers.Embedding(input_dim=max_len, output_dim=embed_dim)\n\n    def call(self, x):\n        max_len = tf.shape(x)[1]\n        positions = tf.range(start=0, limit=max_len, delta=1)\n        return x + self.pos_embedding(positions)\n\nclass MaskingLayer(tf.keras.layers.Layer):\n    def call(self, inputs):\n        frames, non_empty_frame_idxs = inputs\n        mask = tf.math.not_equal(non_empty_frame_idxs, -1)\n        mask = tf.cast(mask, dtype=frames.dtype)\n        return frames * tf.expand_dims(mask, -1)\n\ndef build_inference_model():\n    embed_dim = 192\n    num_heads = 4\n    ff_dim = embed_dim * 2\n    \n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS_FINAL], dtype=tf.float16, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float16, name='non_empty_frame_idxs')\n    \n    x = MaskingLayer(name='input_masking')([frames, non_empty_frame_idxs])\n    x = tf.keras.layers.Dense(embed_dim, use_bias=False, name='stem_conv')(x)\n    x = tf.keras.layers.BatchNormalization(momentum=0.95, name='stem_bn')(x)\n    \n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    \n    x = LearnablePositionalEmbedding(INPUT_SIZE, embed_dim)(x)\n    x = TransformerBlock(embed_dim, num_heads, ff_dim, rate=0.2)(x)\n    \n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    x = Conv1DBlock(embed_dim, kernel_size=17, drop_rate=0.2)(x)\n    \n    x = TransformerBlock(embed_dim, num_heads, ff_dim, rate=0.2)(x)\n    \n    x = tf.keras.layers.GlobalAveragePooling1D()(x)\n    x = tf.keras.layers.Dropout(0.8)(x) \n    outputs = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax', dtype='float32', name='classifier')(x)\n    \n    return tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n\n# Initialize and Load Weights\ntf.keras.backend.clear_session()\nmodel = build_inference_model()\nprint(\"Model initialized. Loading weights...\")\n\ntry:\n    model.load_weights(MODEL_WEIGHTS_PATH)\n    print(f\"Weights successfully loaded from:\\n{MODEL_WEIGHTS_PATH}\")\nexcept Exception as e:\n    print(f\"Failed to load weights. Please verify the path.\\nError: {e}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:35:58.468711Z","iopub.execute_input":"2026-04-19T08:35:58.469514Z","iopub.status.idle":"2026-04-19T08:36:02.627903Z","shell.execute_reply.started":"2026-04-19T08:35:58.469479Z","shell.execute_reply":"2026-04-19T08:36:02.627243Z"}},"outputs":[{"name":"stderr","text":"I0000 00:00:1776587759.444050     117 gpu_device.cc:2019] Created device /job:localhost/replica:0/task:0/device:GPU:0 with 13757 MB memory:  -> device: 0, name: Tesla T4, pci bus id: 0000:00:04.0, compute capability: 7.5\nI0000 00:00:1776587759.450047     117 gpu_device.cc:2019] Created device /job:localhost/replica:0/task:0/device:GPU:1 with 13757 MB memory:  -> device: 1, name: Tesla T4, pci bus id: 0000:00:05.0, compute capability: 7.5\n","output_type":"stream"},{"name":"stdout","text":"Model initialized. Loading weights...\nWeights successfully loaded from:\n/kaggle/input/models/idowuadamo/hybrid-model-toptier/tensorflow2/default/1/hybrid_model_abs.weights.h5\n","output_type":"stream"}],"execution_count":6},{"cell_type":"code","source":"# Initialize LLM\ntry:\n    from kaggle_secrets import UserSecretsClient\n    login(token=UserSecretsClient().get_secret(\"HF_TOKEN\"))\nexcept:\n    if os.environ.get(\"HF_TOKEN\"): login(token=os.environ.get(\"HF_TOKEN\"))\n\nprint(\"Loading Llama 3.2 1B Instruct...\")\nimport torch\nfrom transformers import pipeline\n\nllm = pipeline(\n    \"text-generation\", \n    model=\"meta-llama/Llama-3.2-1B-Instruct\", \n    device_map=\"auto\", \n    torch_dtype=torch.float16,\n    model_kwargs={\"attn_implementation\": \"eager\"} \n)\nprint(\"LLM Loaded successfully.\")\n\n# Initialize MediaPipe Holistic\nimport mediapipe as mp\nmp_holistic = mp.solutions.holistic\nholistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5)\nprint(\"MediaPipe initialized.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:36:06.476986Z","iopub.execute_input":"2026-04-19T08:36:06.477661Z","iopub.status.idle":"2026-04-19T08:36:22.888637Z","shell.execute_reply.started":"2026-04-19T08:36:06.477626Z","shell.execute_reply":"2026-04-19T08:36:22.886923Z"}},"outputs":[{"name":"stdout","text":"Loading Llama 3.2 1B Instruct...\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"config.json:   0%|          | 0.00/877 [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"e17f66a8636143be848d76ada790dd42"}},"metadata":{}},{"name":"stderr","text":"`torch_dtype` is deprecated! Use `dtype` instead!\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"model.safetensors:   0%|          | 0.00/2.47G [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"fcc8d4cf34904f159e7396e09074d019"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"Loading weights:   0%|          | 0/146 [00:00<?, ?it/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"04a52017dcf647a598d8b6b25a64fd6d"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"generation_config.json:   0%|          | 0.00/189 [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"7d860a5c2c1b4e4c855f117f77378a77"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"tokenizer_config.json: 0.00B [00:00, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"a497e5cfbaee43da9939d8f78c076043"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"tokenizer.json: 0.00B [00:00, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"913c8c7a2cef46df8a9594505c9bea5f"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"special_tokens_map.json:   0%|          | 0.00/296 [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"89e7862e98e24b1f98ad9ede98847b52"}},"metadata":{}},{"name":"stdout","text":"LLM Loaded successfully.\nMediaPipe initialized.\n","output_type":"stream"},{"name":"stderr","text":"INFO: Created TensorFlow Lite XNNPACK delegate for CPU.\nWARNING: All log messages before absl::InitializeLog() is called are written to STDERR\nW0000 00:00:1776587783.000400     751 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.037187     751 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.038386     749 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.038730     751 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.039846     748 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.050049     748 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.062740     749 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\nW0000 00:00:1776587783.070540     751 inference_feedback_manager.cc:114] Feedback manager requires a model with a single signature inference. Disabling support for feedback tensors.\n","output_type":"stream"}],"execution_count":7},{"cell_type":"code","source":"def extract_landmarks(results):\n    \"\"\"Extract 543 landmarks per frame (Lips, Left Hand, Pose, Right Hand).\"\"\"\n    landmarks = []\n    \n    if results.face_landmarks:\n        landmarks.extend([[lm.x, lm.y, lm.z] for lm in results.face_landmarks.landmark])\n    else:\n        landmarks.extend([[float('nan')] * 3] * 468)\n\n    if results.left_hand_landmarks:\n        landmarks.extend([[lm.x, lm.y, lm.z] for lm in results.left_hand_landmarks.landmark])\n    else:\n        landmarks.extend([[float('nan')] * 3] * 21)\n\n    if results.pose_landmarks:\n        landmarks.extend([[lm.x, lm.y, lm.z] for lm in results.pose_landmarks.landmark])\n    else:\n        landmarks.extend([[float('nan')] * 3] * 33)\n\n    if results.right_hand_landmarks:\n        landmarks.extend([[lm.x, lm.y, lm.z] for lm in results.right_hand_landmarks.landmark])\n    else:\n        landmarks.extend([[float('nan')] * 3] * 21)\n\n    return np.array(landmarks, dtype=np.float32)\n\ndef refine_to_sentence(sign_sequence):\n    \"\"\"Converts ASL glosses to English using Llama 3.2.\"\"\"\n    if not sign_sequence:\n        return \"No signs detected.\"\n        \n    glosses_str = \" \".join(sign_sequence)\n    messages = [\n        {\"role\": \"system\", \"content\": \"You are a fast, accurate ASL-to-English translation engine. Convert the provided ASL gloss sequence into a fluent, grammatically correct English sentence. Do not add explanations. Output ONLY the final translated sentence.\"},\n        {\"role\": \"user\", \"content\": f\"Glosses: {glosses_str}\"}\n    ]\n    \n    prompt = llm.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)\n    out = llm(prompt, max_new_tokens=64, do_sample=False, return_full_text=False)[0]['generated_text']\n    return out.strip()\n\ndef process_video_pipeline(video_path):\n    \"\"\"End-to-End inference processing.\"\"\"\n    start_time = time.time()\n    if not video_path:\n        return \"No video provided.\", \"\", \"N/A\", None, \"0.0s / 0.0 FPS\"\n\n    # Video Parsing & MediaPipe\n    cap = cv2.VideoCapture(video_path)\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    frames_lms = []\n    \n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret: break\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        results = holistic.process(frame_rgb)\n        frames_lms.append(extract_landmarks(results))\n    cap.release()\n    \n    if len(frames_lms) < 10:\n        return \"Video too short.\", \"Video too short.\", \"N/A\", None, \"0.0s\"\n        \n    mp_time = time.time()\n    \n    # Tensor Batching & Preprocessing\n    data = np.array(frames_lms, dtype=np.float32)\n    batch_frames, batch_idxs = [], []\n    window_step = 32  \n    \n    for start in range(0, max(1, len(data) - INPUT_SIZE + 1), window_step):\n        window = data[start:start + INPUT_SIZE]\n        if len(window) < INPUT_SIZE:\n            pad_len = INPUT_SIZE - len(window)\n            window = np.concatenate((window, np.zeros((pad_len, 543, 3))), axis=0)\n            \n        p_data, n_idxs = preprocess_layer(window)\n        batch_frames.append(p_data)\n        batch_idxs.append(n_idxs)\n        \n    prep_time = time.time()\n        \n    # TF GPU Inference\n    # Convert lists to dense tensors for prediction\n    preds_raw = model.predict({\n        'frames': tf.convert_to_tensor(batch_frames, dtype=tf.float16), \n        'non_empty_frame_idxs': tf.convert_to_tensor(batch_idxs, dtype=tf.float16)\n    }, batch_size=32, verbose=0)\n    \n    # Gloss Aggregation\n    CONFIDENCE_THRESH = 0.4\n    unique_seq = []\n    conf_log = []\n    \n    for p in preds_raw:\n        pred_idx = p.argmax()\n        conf = p[pred_idx]\n        sign = ORD2SIGN.get(pred_idx, 'unknown')\n        \n        if sign != \"unknown\" and conf > CONFIDENCE_THRESH:\n            if not unique_seq or sign != unique_seq[-1]:\n                unique_seq.append(sign)\n                conf_log.append(f\"{sign} ({conf:.1%})\")\n                \n    inf_time = time.time()\n    \n    if not unique_seq:\n        return \"No confident signs recognized.\", \"None\", \"N/A\", None, f\"Processed in {inf_time - start_time:.2f}s\"\n        \n    raw_glosses = \" \".join(unique_seq)\n    formatted_confs = \" | \".join(conf_log)\n    \n    # LLM Refinement\n    refined_sentence = refine_to_sentence(unique_seq)\n    llm_time = time.time()\n    \n    # Text-to-Speech\n    audio_path = 'output.mp3'\n    try:\n        tts = gTTS(refined_sentence, lang='en')\n        tts.save(audio_path)\n    except:\n        audio_path = None\n        \n    end_time = time.time()\n    \n    total_time = end_time - start_time\n    process_fps = total_frames / total_time if total_time > 0 else 0\n    stats = (f\"Total Latency: {total_time:.2f}s | \"\n             f\"Speed: {process_fps:.1f} FPS\\n\"\n             f\"(Breakdown - MediaPipe: {mp_time-start_time:.2f}s, TF: {inf_time-prep_time:.2f}s, LLM: {llm_time-inf_time:.2f}s)\")\n             \n    return raw_glosses, formatted_confs, refined_sentence, audio_path, stats\n\n# Gradio UI\ncustom_css = \"\"\"\n#large_text textarea { font-size: 28px !important; font-weight: 700; color: #1a202c; }\n\"\"\"\nwith gr.Blocks(theme=gr.themes.Soft(primary_hue=\"indigo\"), css=custom_css) as demo:\n    gr.Markdown(\"# Real-Time ASL Translation Pipeline\")\n    \n    with gr.Row():\n        with gr.Column(scale=5):\n            gr.Markdown(\"### 1. Video Input\")\n            vid_input = gr.Video(sources=[\"webcam\", \"upload\"], label=\"Record or Upload\", include_audio=False)\n            btn_translate = gr.Button(\"🚀 Run Full Inference Pipeline\", variant=\"primary\", size=\"lg\")\n            \n            gr.Markdown(\"### 2. Pipeline Diagnostics\")\n            out_stats = gr.Textbox(label=\"Processing Time & FPS\", interactive=False, lines=2)\n            out_confs = gr.Textbox(label=\"Per-Gloss Confidence Scores\", interactive=False)\n            \n        with gr.Column(scale=5):\n            gr.Markdown(\"### 3. Translation Output\")\n            out_sentence = gr.Textbox(label=\"Final English Sentence\", elem_id=\"large_text\", lines=3, interactive=False)\n            out_audio = gr.Audio(label=\"Spoken Translation\", autoplay=True)\n            out_glosses = gr.Textbox(label=\"Raw Gloss Sequence (Tier 1)\", interactive=False)\n\n    btn_translate.click(\n        fn=process_video_pipeline, inputs=vid_input,\n        outputs=[out_glosses, out_confs, out_sentence, out_audio, out_stats]\n    )\n\ndemo.launch(share=True, debug=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T08:36:45.524607Z","iopub.execute_input":"2026-04-19T08:36:45.524942Z","execution_failed":"2026-04-19T08:39:36.767Z"}},"outputs":[{"name":"stderr","text":"/tmp/ipykernel_117/3162858124.py:139: DeprecationWarning: The 'theme' parameter in the Blocks constructor will be removed in Gradio 6.0. You will need to pass 'theme' to Blocks.launch() instead.\n  with gr.Blocks(theme=gr.themes.Soft(primary_hue=\"indigo\"), css=custom_css) as demo:\n/tmp/ipykernel_117/3162858124.py:139: DeprecationWarning: The 'css' parameter in the Blocks constructor will be removed in Gradio 6.0. You will need to pass 'css' to Blocks.launch() instead.\n  with gr.Blocks(theme=gr.themes.Soft(primary_hue=\"indigo\"), css=custom_css) as demo:\n","output_type":"stream"},{"name":"stdout","text":"* Running on local URL:  http://127.0.0.1:7860\n* Running on public URL: https://2461376ecc1ded9b25.gradio.live\n\nThis share link expires in 1 week. For free permanent hosting and GPU upgrades, run `gradio deploy` from the terminal in the working directory to deploy to Hugging Face Spaces (https://huggingface.co/spaces)\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"<IPython.core.display.HTML object>","text/html":"<div><iframe src=\"https://2461376ecc1ded9b25.gradio.live\" width=\"100%\" height=\"500\" allow=\"autoplay; camera; microphone; clipboard-read; clipboard-write;\" frameborder=\"0\" allowfullscreen></iframe></div>"},"metadata":{}}],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}