{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset , DataLoader \nimport pandas as pd\nimport numpy as np\nimport json \nimport os \nimport glob\nfrom sklearn.model_selection import train_test_split\n\n#pathes \nROOT_DIR=\"/kaggle/input/competitions/asl-signs\"\nCSV_PATH= os.path.join(ROOT_DIR,\"train.csv\")\nJSON_PATH = os.path.join(ROOT_DIR,\"sign_to_prediction_index_map.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T17:13:18.834908Z","iopub.execute_input":"2026-04-24T17:13:18.835167Z","iopub.status.idle":"2026-04-24T17:13:28.637661Z","shell.execute_reply.started":"2026-04-24T17:13:18.835135Z","shell.execute_reply":"2026-04-24T17:13:28.636986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(CSV_PATH)\nfirst_row = df.iloc[0]\nsample_path = os.path.join(ROOT_DIR, first_row['path'])\nsample_parquet = pd.read_parquet(sample_path)\n\nprint(\"Parquet Head:\")\nprint(sample_parquet.head())\n\nunique_landmarks = sample_parquet['landmark_index'].unique()\nprint(f\"\\nUnique landmarks count: {len(unique_landmarks)}\") \n\nprint(\"\\n--- Groups Inspection ---\")\nfor type_name in sample_parquet['type'].unique():\n    group_data = sample_parquet[sample_parquet['type'] == type_name]['landmark_index']\n    first_idx = group_data.min() \n    last_idx = group_data.max()\n    \n    print(f\"Group: {type_name} | Start: {first_idx} | End: {last_idx}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T17:15:07.682425Z","iopub.execute_input":"2026-04-24T17:15:07.683246Z","iopub.status.idle":"2026-04-24T17:15:08.258547Z","shell.execute_reply.started":"2026-04-24T17:15:07.683182Z","shell.execute_reply":"2026-04-24T17:15:08.257767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"csv_files = glob.glob('/kaggle/input/**/train.csv', recursive=True)\n\nif not csv_files:\n    print(\"train.csv not found\")\nelse:\n    csv_path = csv_files[0]\n    root_dir = os.path.dirname(csv_path) \n    \n    print(f\"Found CSV at: {csv_path}\")\n    print(f\"Root Directory is: {root_dir}\")\n    \n    train_df = pd.read_csv(csv_path)\n    first_parquet = os.path.join(root_dir, train_df['path'].iloc[0])\n    sample_data = pd.read_parquet(first_parquet)\n    \n    print(sample_data['frame'].unique()[:10])\n    print(\"\\n--- Success! First 10 frames ---\")\n    print(sample_data['frame'].head(10).tolist())\n    print(f\"\\nIs it sorted? {sample_data['frame'].is_monotonic_increasing}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T15:08:40.704382Z","iopub.execute_input":"2026-04-24T15:08:40.704741Z","iopub.status.idle":"2026-04-24T15:11:45.354107Z","shell.execute_reply.started":"2026-04-24T15:08:40.704713Z","shell.execute_reply":"2026-04-24T15:11:45.352856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GoogleSignDataset(Dataset):\n    def __init__(self, df, ROOT_DIR, sign_map, n_frames=30, mode='train'):\n        self.df = df\n        self.root_dir = root_dir\n        self.sign_map = sign_map\n        self.n_frames = n_frames\n        self.mode = mode\n\n        # Specific landmark indices for facial features (Lips, Eyes, Brows)\n        self.FACE_IDXS = [\n            # Lips\n            61, 146, 91, 181, 84, 17, 314, 405, 321, 375, 291, 409, 270, 269, 267, 0, 37, 39, 40, 185,\n            # Left Eye & Eyebrow\n            33, 7, 163, 144, 145, 153, 154, 155, 133, 46, 53, 52, 65, 55, 70, 63, 105, 66, 107,\n            # Right Eye & Eyebrow\n            263, 249, 390, 373, 374, 380, 381, 382, 362, 276, 283, 282, 295, 285, 300, 293, 334, 296, 336\n        ]\n        \n        # Standard landmark ranges for Hands and Pose\n        self.L_HAND_IDXS = list(range(21))\n        self.R_HAND_IDXS = list(range(21))\n        self.POSE_IDXS = list(range(33))\n\n    def __len__(self):\n        # Returns the total number of samples in the dataframe\n        return len(self.df)\n\n    def load_video(self, path):\n        # Construct full path and load the parquet file\n        full_path = os.path.join(self.root_dir, path)\n        data = pd.read_parquet(full_path)\n        \n        # Get unique frames and maintain chronological order\n        unique_frames = sorted(data['frame'].unique())\n        n_frames_in_video = len(unique_frames)\n        \n        # Initialize empty arrays for separate data streams (Face vs. Body)\n        # Body stream combines Left Hand, Right Hand, and Pose (21+21+33 = 75 points)\n        face_data = np.zeros((n_frames_in_video, len(self.FACE_IDXS), 3), dtype=np.float32)\n        body_data = np.zeros((n_frames_in_video, 75, 3), dtype=np.float32)\n        \n        for i, frame in enumerate(unique_frames):\n            frame_data = data[data['frame'] == frame]\n            \n            # 1. Process Face Stream: Filter and reindex specific landmarks\n            face = frame_data[frame_data['type'] == 'face']\n            if not face.empty:\n                f_vals = face.set_index('landmark_index').reindex(self.FACE_IDXS)[['x', 'y', 'z']].values\n                face_data[i] = np.nan_to_num(f_vals)\n\n            # 2. Process Body Stream: Concatenate Hands and Pose data\n            current_pos = 0\n            for t, idxs in [('left_hand', self.L_HAND_IDXS), \n                            ('right_hand', self.R_HAND_IDXS), \n                            ('pose', self.POSE_IDXS)]:\n                group = frame_data[frame_data['type'] == t]\n                if not group.empty:\n                    b_vals = group.set_index('landmark_index').reindex(idxs)[['x', 'y', 'z']].values\n                    body_data[i, current_pos : current_pos + len(idxs)] = np.nan_to_num(b_vals)\n                current_pos += len(idxs)\n                \n        return face_data, body_data\n\n    def resample(self, data, size):\n        # Standardizes sequence length to exactly 'size' frames\n        # If longer: downsample using linear interpolation of indices\n        # If shorter: pad using edge-case repetition\n        if len(data) >= size:\n            indices = np.linspace(0, len(data) - 1, size).astype(int)\n        else:\n            indices = np.pad(np.arange(len(data)), (0, size - len(data)), mode='edge')\n        return data[indices]\n\n  def augment_landmarks(self, x):\n        noise = np.random.normal(0, 0.002, x.shape)\n        x = x + noise\n      \n        scale = np.random.uniform(0.9, 1.1)\n        x = x * scale\n        \n        shift = np.random.uniform(-0.05, 0.05, size=(x.shape[-1]))\n        x = x + shift\n      \n        angle = np.deg2rad(np.random.uniform(-5, 5))\n        rot_mat = np.array([\n            [np.cos(angle), -np.sin(angle), 0],\n            [np.sin(angle),  np.cos(angle), 0],\n            [0, 0, 1]\n        ])\n        x = np.dot(x.reshape(-1, 3), rot_mat.T).reshape(x.shape)\n        \n        return x\n\n    def __getitem__(self, index):\n        # Retrieve row info from dataframe\n        row = self.df.iloc[index]\n        \n        # Load raw Face and Body streams\n        face_v, body_v = self.load_video(row['path'])\n\n        \n        if self.mode == 'train':\n          face_v = self.augment_landmarks(face_v)\n          body_v = self.augment_landmarks(body_v)\n        \n        # Standardize temporal dimension (T) for both streams to match n_frames\n        face_v = self.resample(face_v, self.n_frames)\n        body_v = self.resample(body_v, self.n_frames)\n        \n        # Map sign string to numerical label\n        label = self.sign_map[row['sign']]\n\n        # Return a dictionary of Tensors for the Multimodal Model\n        return {\n            'face': torch.tensor(face_v, dtype=torch.float32), # Output shape: (T, 58, 3)\n            'body': torch.tensor(body_v, dtype=torch.float32), # Output shape: (T, 75, 3)\n            'label': torch.tensor(label, dtype=torch.long)\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T17:15:18.892997Z","iopub.execute_input":"2026-04-24T17:15:18.893320Z","iopub.status.idle":"2026-04-24T17:15:18.903963Z","shell.execute_reply.started":"2026-04-24T17:15:18.893292Z","shell.execute_reply":"2026-04-24T17:15:18.902697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(JSON_PATH, 'r') as f:\n    sign_map = json.load(f)\n\ntrain_df, temp_df = train_test_split(\n    df, \n    test_size=0.2, \n    random_state=42, \n    stratify=df['sign']\n)\n\nval_df, test_df = train_test_split(\n    temp_df, \n    test_size=0.5, \n    random_state=42, \n    stratify=temp_df['sign']\n)\ntrain_dataset = GoogleSignDataset(train_df, ROOT_DIR, sign_map, mode='train')\nval_dataset   = GoogleSignDataset(val_df, ROOT_DIR, sign_map,mode='val')\ntest_dataset  = GoogleSignDataset(test_df, ROOT_DIR, sign_map)\n\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=2)\nval_loader   = DataLoader(val_dataset, batch_size=32, shuffle=False,  num_workers=2)\ntest_loader  = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=2)\n\nprint(f\"Train size: {len(train_df)}\")\nprint(f\"Val size: {len(val_df)}\")\nprint(f\"Test size: {len(test_df)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"model_architecture","metadata":{}},{"cell_type":"code","source":"class ASLSignModel(nn.Module):\n\n    def __init__(self, num_classes=250, embedding_dim=256, llm_hidden_size=896):\n\n        super(ASLSignModel, self).__init__()\n\n        # 1. Face Stream (Linear + LSTM)\n\n        self.face_fc = nn.Linear(174, embedding_dim)\n\n        self.face_lstm = nn.LSTM(embedding_dim, embedding_dim, batch_first=True, num_layers=2, bidirectional=True)\n\n        # 2. Body Stream (Linear + LSTM)\n\n        self.body_fc = nn.Linear(225, embedding_dim)\n\n        self.body_lstm = nn.LSTM(embedding_dim, embedding_dim, batch_first=True, num_layers=2, bidirectional=True)\n\n        # 3. Projector (The Bridge to Qwen)\n\n        self.qwen_projector = nn.Linear(embedding_dim * 2, llm_hidden_size)\n\n        # 4. Classifier (For Training)\n\n        self.classifier = nn.Linear(llm_hidden_size, num_classes)\n\n\n    def forward(self, face, body):\n\n        # 1. Processing Face\n\n        f = torch.relu(self.face_fc(face.view(face.size(0), face.size(1), -1)))\n\n        f_out, (h_f, _) = self.face_lstm(f)\n\n        feat_face = torch.cat((h_f[-2,:,:], h_f[-1,:,:]), dim=-1) # (Batch, 512)\n\n        \n\n        # 2. Processing Body\n\n        b = torch.relu(self.body_fc(body.view(body.size(0), body.size(1), -1)))\n\n        b_out, (h_b, _) = self.body_lstm(b)\n\n        feat_body = torch.cat((h_b[-2,:,:], h_b[-1,:,:]), dim=-1) # (Batch, 512)\n\n        # 3. Average Pooling \n\n        combined = (feat_face + feat_body) / 2 # (Batch, 512)\n\n        # 4. Project to Qwen space\n\n        qwen_embedding = self.qwen_projector(combined) # (Batch, 896)\n\n        # 5. Output for classification\n\n        logits = self.classifier(qwen_embedding)\n\n        return logits, qwen_embedding","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = ASLSignModel(num_classes=n_classes).to(device)\n\n# Hyperparameters\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\ncriterion = nn.CrossEntropyLoss()\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3)\n\nepochs = 100\nbest_val_acc = 0.0\n\nprint(f\"Device: {device}\")\n\nfor epoch in range(1, epochs + 1):\n    start_time = time.time()\n    \n    # Training Phase\n    model.train()\n    train_loss = 0.0\n    train_correct = 0\n    total_train = 0\n    \n    for face, body, labels in train_loader:\n        face, body, labels = face.to(device), body.to(device), labels.to(device)\n        \n        optimizer.zero_grad()\n        logits, _ = model(face, body)\n        loss = criterion(logits, labels)\n        \n        loss.backward()\n        torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n        optimizer.step()\n        \n        train_loss += loss.item() * face.size(0)\n        train_correct += (torch.argmax(logits, dim=1) == labels).sum().item()\n        total_train += labels.size(0)\n\n    # Validation Phase\n    model.eval()\n    val_loss = 0.0\n    val_correct = 0\n    total_val = 0\n    \n    with torch.no_grad():\n        for face, body, labels in val_loader:\n            face, body, labels = face.to(device), body.to(device), labels.to(device)\n            \n            logits, _ = model(face, body)\n            loss = criterion(logits, labels)\n            \n            val_loss += loss.item() * face.size(0)\n            val_correct += (torch.argmax(logits, dim=1) == labels).sum().item()\n            total_val += labels.size(0)\n\n    # Calculations\n    avg_train_loss = train_loss / total_train\n    avg_train_acc = (train_correct / total_train) * 100\n    avg_val_loss = val_loss / total_val\n    avg_val_acc = (val_correct / total_val) * 100\n    \n    current_lr = optimizer.param_groups[0]['lr']\n    scheduler.step(avg_val_loss)\n    duration = time.time() - start_time\n\n    # Output Metrics\n    print(f\"Epoch {epoch:03d} | Time: {duration:.1f}s | LR: {current_lr:.6f}\")\n    print(f\"TRAIN - Loss: {avg_train_loss:.4f}, Acc: {avg_train_acc:.2f}%\")\n    print(f\"VAL   - Loss: {avg_val_loss:.4f}, Acc: {avg_val_acc:.2f}%\")\n\n    # Save Best Model\n    if avg_val_acc > best_val_acc:\n        best_val_acc = avg_val_acc\n        torch.save(model.state_dict(), 'best_asl_model.pt_2')\n        print(f\"New Best Accuracy: {best_val_acc:.2f}%\")\n    \n    print(\"-\" * 30)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nimport numpy as np\nimport json\nimport os\nimport glob\nimport time\nfrom sklearn.model_selection import train_test_split\n\n# 1. Paths and Data Loading\nROOT_DIR = \"/kaggle/input/competitions/asl-signs\"\nCSV_PATH = os.path.join(ROOT_DIR, \"train.csv\")\nJSON_PATH = os.path.join(ROOT_DIR, \"sign_to_prediction_index_map.json\")\n\ndf = pd.read_csv(CSV_PATH)\nwith open(JSON_PATH, 'r') as f:\n    sign_map = json.load(f)\n\nn_classes = len(sign_map)\n\n# 2. Dataset Class\nclass GoogleSignDataset(Dataset):\n    def __init__(self, df, root_dir, sign_map, n_frames=30, mode='train'):\n        self.df = df\n        self.root_dir = root_dir\n        self.sign_map = sign_map\n        self.n_frames = n_frames\n        self.mode = mode\n        self.cache = {}\n        \n        self.FACE_IDXS = [\n            61, 146, 91, 181, 84, 17, 314, 405, 321, 375, 291, 409, 270, 269, 267, 0, 37, 39, 40, 185,\n            33, 7, 163, 144, 145, 153, 154, 155, 133, 46, 53, 52, 65, 55, 70, 63, 105, 66, 107,\n            263, 249, 390, 373, 374, 380, 381, 382, 362, 276, 283, 282, 295, 285, 300, 293, 334, 296, 336\n        ]\n        self.L_HAND_IDXS = list(range(21))\n        self.R_HAND_IDXS = list(range(21))\n        self.POSE_IDXS = list(range(33))\n\n    def __len__(self):\n        return len(self.df)\n\n    def load_video(self, path):\n        if path in self.cache:\n            return self.cache[path]\n            \n        full_path = os.path.join(self.root_dir, path)\n        data = pd.read_parquet(full_path)\n        unique_frames = sorted(data['frame'].unique())\n        n_frames_in_video = len(unique_frames)\n        \n        face_data = np.zeros((n_frames_in_video, len(self.FACE_IDXS), 3), dtype=np.float32)\n        body_data = np.zeros((n_frames_in_video, 75, 3), dtype=np.float32)\n        \n        for i, frame in enumerate(unique_frames):\n            frame_data = data[data['frame'] == frame]\n            \n            face = frame_data[frame_data['type'] == 'face']\n            if not face.empty:\n                f_vals = face.set_index('landmark_index').reindex(self.FACE_IDXS)[['x', 'y', 'z']].values\n                face_data[i] = np.nan_to_num(f_vals)\n\n            current_pos = 0\n            for t, idxs in [('left_hand', self.L_HAND_IDXS), ('right_hand', self.R_HAND_IDXS), ('pose', self.POSE_IDXS)]:\n                group = frame_data[frame_data['type'] == t]\n                if not group.empty:\n                    b_vals = group.set_index('landmark_index').reindex(idxs)[['x', 'y', 'z']].values\n                    body_data[i, current_pos : current_pos + len(idxs)] = np.nan_to_num(b_vals)\n                current_pos += len(idxs)\n        \n        self.cache[path] = (face_data, body_data)\n        \n        return face_data, body_data\n        \n    def resample(self, data, size):\n        if len(data) >= size:\n            indices = np.linspace(0, len(data) - 1, size).astype(int)\n        else:\n            indices = np.pad(np.arange(len(data)), (0, size - len(data)), mode='edge')\n        return data[indices]\n\n    def augment_landmarks(self, x):\n        noise = np.random.normal(0, 0.002, x.shape)\n        x = x + noise\n        scale = np.random.uniform(0.9, 1.1)\n        x = x * scale\n        shift = np.random.uniform(-0.05, 0.05, size=(x.shape[-1]))\n        x = x + shift\n        angle = np.deg2rad(np.random.uniform(-5, 5))\n        rot_mat = np.array([\n            [np.cos(angle), -np.sin(angle), 0],\n            [np.sin(angle),  np.cos(angle), 0],\n            [0, 0, 1]\n        ])\n        x = np.dot(x.reshape(-1, 3), rot_mat.T).reshape(x.shape)\n        return x\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        face_v, body_v = self.load_video(row['path'])\n        \n        if self.mode == 'train':\n            face_v = self.augment_landmarks(face_v)\n            body_v = self.augment_landmarks(body_v)\n        \n        face_v = self.resample(face_v, self.n_frames).reshape(self.n_frames, -1)\n        body_v = self.resample(body_v, self.n_frames).reshape(self.n_frames, -1)\n        \n        face_v = (face_v - face_v.mean()) / (face_v.std() + 1e-6)\n        body_v = (body_v - body_v.mean()) / (body_v.std() + 1e-6)\n        \n        label = self.sign_map[row['sign']]\n        return torch.tensor(face_v, dtype=torch.float32), torch.tensor(body_v, dtype=torch.float32), torch.tensor(label, dtype=torch.long)\n\n# 3. Model Architecture\nclass ASLSignModel(nn.Module):\n    def __init__(self, num_classes=250, embedding_dim=256, llm_hidden_size=896):\n        super(ASLSignModel, self).__init__()\n        self.face_fc = nn.Linear(174, embedding_dim)\n        self.face_lstm = nn.LSTM(embedding_dim, embedding_dim, batch_first=True, num_layers=2, bidirectional=True)\n        self.body_fc = nn.Linear(225, embedding_dim)\n        self.body_lstm = nn.LSTM(embedding_dim, embedding_dim, batch_first=True, num_layers=2, bidirectional=True)\n        self.qwen_projector = nn.Linear(embedding_dim * 2, llm_hidden_size)\n        self.classifier = nn.Linear(llm_hidden_size, num_classes)\n\n    def forward(self, face, body):\n        f = torch.relu(self.face_fc(face))\n        f_out, (h_f, _) = self.face_lstm(f)\n        feat_face = torch.cat((h_f[-2,:,:], h_f[-1,:,:]), dim=-1)\n        \n        b = torch.relu(self.body_fc(body))\n        b_out, (h_b, _) = self.body_lstm(b)\n        feat_body = torch.cat((h_b[-2,:,:], h_b[-1,:,:]), dim=-1)\n        \n        combined = (feat_face + feat_body) / 2\n        qwen_embedding = self.qwen_projector(combined)\n        logits = self.classifier(qwen_embedding)\n        return logits, qwen_embedding\n\n# 4. Data Preparation\ntrain_df, temp_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df['sign'])\nval_df, test_df = train_test_split(temp_df, test_size=0.5, random_state=42, stratify=temp_df['sign'])\n\ntrain_dataset = GoogleSignDataset(train_df, ROOT_DIR, sign_map, mode='train')\nval_dataset   = GoogleSignDataset(val_df, ROOT_DIR, sign_map, mode='val')\n\ntrain_loader = DataLoader(train_dataset, batch_size=64, shuffle=True, num_workers=4,pin_memory=True,persistent_workers=True)\nval_loader   = DataLoader(val_dataset, batch_size=64, shuffle=False, num_workers=4,pin_memory=True,persistent_workers=True)\n\n# 5. Training Loop\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = ASLSignModel(num_classes=n_classes).to(device)\n\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\ncriterion = nn.CrossEntropyLoss()\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3)\n\nepochs = 100\nbest_val_acc = 0.0\n\nfor epoch in range(1, epochs + 1):\n    start_time = time.time()\n    model.train()\n    train_loss, train_correct, total_train = 0.0, 0, 0\n    \n    for face, body, labels in train_loader:\n        face, body, labels = face.to(device), body.to(device), labels.to(device)\n        optimizer.zero_grad()\n        logits, _ = model(face, body)\n        loss = criterion(logits, labels)\n        loss.backward()\n        torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n        optimizer.step()\n        \n        train_loss += loss.item() * face.size(0)\n        train_correct += (torch.argmax(logits, dim=1) == labels).sum().item()\n        total_train += labels.size(0)\n\n    model.eval()\n    val_loss, val_correct, total_val = 0.0, 0, 0\n    with torch.no_grad():\n        for face, body, labels in val_loader:\n            face, body, labels = face.to(device), body.to(device), labels.to(device)\n            logits, _ = model(face, body)\n            loss = criterion(logits, labels)\n            val_loss += loss.item() * face.size(0)\n            val_correct += (torch.argmax(logits, dim=1) == labels).sum().item()\n            total_val += labels.size(0)\n\n    avg_train_loss, avg_train_acc = train_loss / total_train, (train_correct / total_train) * 100\n    avg_val_loss, avg_val_acc = val_loss / total_val, (val_correct / total_val) * 100\n    \n    current_lr = optimizer.param_groups[0]['lr']\n    scheduler.step(avg_val_loss)\n    \n    print(f\"Epoch {epoch:03d} | Time: {time.time()-start_time:.1f}s | LR: {current_lr:.6f}\")\n    print(f\"TRAIN - Loss: {avg_train_loss:.4f}, Acc: {avg_train_acc:.2f}%\")\n    print(f\"VAL   - Loss: {avg_val_loss:.4f}, Acc: {avg_val_acc:.2f}%\")\n\n    if avg_val_acc > best_val_acc:\n        best_val_acc = avg_val_acc\n        torch.save(model.state_dict(), 'best_asl_model.pt_2')\n        print(f\"New Best Accuracy: {best_val_acc:.2f}%\")\n    print(\"-\" * 30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T18:33:01.067405Z","iopub.execute_input":"2026-04-24T18:33:01.068142Z","iopub.status.idle":"2026-04-24T18:41:46.022778Z","shell.execute_reply.started":"2026-04-24T18:33:01.068105Z","shell.execute_reply":"2026-04-24T18:41:46.019964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nimport numpy as np\nimport json\nimport os\nimport time\nfrom sklearn.model_selection import train_test_split\n\n# 1. Paths\nROOT_DIR = \"/kaggle/input/competitions/asl-signs\"\nCSV_PATH = os.path.join(ROOT_DIR, \"train.csv\")\nJSON_PATH = os.path.join(ROOT_DIR, \"sign_to_prediction_index_map.json\")\n\ndf = pd.read_csv(CSV_PATH)\nwith open(JSON_PATH, 'r') as f:\n    sign_map = json.load(f)\nn_classes = len(sign_map)\n\n# 2. Optimized Dataset Class\nclass GoogleSignDataset(Dataset):\n    def __init__(self, df, root_dir, sign_map, n_frames=30, mode='train'):\n        self.df = df\n        self.root_dir = root_dir\n        self.sign_map = sign_map\n        self.n_frames = n_frames\n        self.mode = mode\n        \n        # Landmarks indices\n        self.FACE_IDXS = [61, 146, 91, 181, 84, 17, 314, 405, 321, 375, 291, 409, 270, 269, 267, 0, 37, 39, 40, 185, 33, 7, 163, 144, 145, 153, 154, 155, 133, 46, 53, 52, 65, 55, 70, 63, 105, 66, 107, 263, 249, 390, 373, 374, 380, 381, 382, 362, 276, 283, 282, 295, 285, 300, 293, 334, 296, 336]\n        self.L_HAND_IDXS = list(range(21))\n        self.R_HAND_IDXS = list(range(21))\n        self.POSE_IDXS = list(range(33))\n\n    def __len__(self):\n        return len(self.df)\n\n    def load_video(self, path):\n        full_path = os.path.join(self.root_dir, path)\n        # Read only necessary columns and cast to float32\n        data = pd.read_parquet(full_path, columns=['frame', 'type', 'landmark_index', 'x', 'y', 'z'])\n        \n        unique_frames = sorted(data['frame'].unique())\n        n_frames_in_video = len(unique_frames)\n        \n        face_data = np.zeros((n_frames_in_video, len(self.FACE_IDXS), 3), dtype=np.float32)\n        body_data = np.zeros((n_frames_in_video, 75, 3), dtype=np.float32)\n        \n        for i, frame in enumerate(unique_frames):\n            frame_data = data[data['frame'] == frame]\n            \n            # Face\n            face = frame_data[frame_data['type'] == 'face']\n            if not face.empty:\n                face_data[i] = np.nan_to_num(face.set_index('landmark_index').reindex(self.FACE_IDXS)[['x', 'y', 'z']].values)\n\n            # Body (Hands + Pose)\n            current_pos = 0\n            for t, idxs in [('left_hand', self.L_HAND_IDXS), ('right_hand', self.R_HAND_IDXS), ('pose', self.POSE_IDXS)]:\n                group = frame_data[frame_data['type'] == t]\n                if not group.empty:\n                    body_data[i, current_pos : current_pos + len(idxs)] = np.nan_to_num(group.set_index('landmark_index').reindex(idxs)[['x', 'y', 'z']].values)\n                current_pos += len(idxs)\n                \n        return face_data, body_data\n\n    def resample(self, data, size):\n        if len(data) >= size:\n            indices = np.linspace(0, len(data) - 1, size).astype(int)\n        else:\n            indices = np.pad(np.arange(len(data)), (0, size - len(data)), mode='edge')\n        return data[indices]\n\n    def augment_landmarks(self, x):\n        noise = np.random.normal(0, 0.002, x.shape)\n        x = x + noise\n        scale = np.random.uniform(0.9, 1.1)\n        x = x * scale\n        return x\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        face_v, body_v = self.load_video(row['path'])\n        \n        if self.mode == 'train':\n            face_v = self.augment_landmarks(face_v)\n            body_v = self.augment_landmarks(body_v)\n        \n        # Standardize and Flatten for Linear layer\n        face_v = self.resample(face_v, self.n_frames).reshape(self.n_frames, -1)\n        body_v = self.resample(body_v, self.n_frames).reshape(self.n_frames, -1)\n        \n        # Normalize\n        face_v = (face_v - np.mean(face_v)) / (np.std(face_v) + 1e-6)\n        body_v = (body_v - np.mean(body_v)) / (np.std(body_v) + 1e-6)\n        \n        label = self.sign_map[row['sign']]\n        return torch.tensor(face_v, dtype=torch.float32), torch.tensor(body_v, dtype=torch.float32), torch.tensor(label, dtype=torch.long)\n\n# 3. Model\nclass ASLSignModel(nn.Module):\n    def __init__(self, num_classes=250, embedding_dim=256, llm_hidden_size=896):\n        super(ASLSignModel, self).__init__()\n        self.face_fc = nn.Linear(174, embedding_dim)\n        self.face_lstm = nn.LSTM(embedding_dim, embedding_dim, batch_first=True, num_layers=2, bidirectional=True)\n        self.body_fc = nn.Linear(225, embedding_dim)\n        self.body_lstm = nn.LSTM(embedding_dim, embedding_dim, batch_first=True, num_layers=2, bidirectional=True)\n        self.qwen_projector = nn.Linear(embedding_dim * 2, llm_hidden_size)\n        self.classifier = nn.Linear(llm_hidden_size, num_classes)\n\n    def forward(self, face, body):\n        f = torch.relu(self.face_fc(face))\n        _, (h_f, _) = self.face_lstm(f)\n        feat_face = torch.cat((h_f[-2], h_f[-1]), dim=-1)\n        \n        b = torch.relu(self.body_fc(body))\n        _, (h_b, _) = self.body_lstm(b)\n        feat_body = torch.cat((h_b[-2], h_b[-1]), dim=-1)\n        \n        combined = (feat_face + feat_body) / 2\n        logits = self.classifier(self.qwen_projector(combined))\n        return logits\n\n# 4. Preparation\ntrain_df, val_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df['sign'])\n\ntrain_dataset = GoogleSignDataset(train_df, ROOT_DIR, sign_map, mode='train')\nval_dataset   = GoogleSignDataset(val_df, ROOT_DIR, sign_map, mode='val')\n\n# Increase batch size and reduce workers for stability\ntrain_loader = DataLoader(train_dataset, batch_size=128, shuffle=True, num_workers=0)\nval_loader   = DataLoader(val_dataset, batch_size=128, shuffle=False, num_workers=0)\n\n# 5. Training\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = ASLSignModel(num_classes=n_classes).to(device)\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\ncriterion = nn.CrossEntropyLoss()\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3)\n\nprint(f\"Starting training on {device}...\")\n\nfor epoch in range(1, 11): # Start with 10 epochs to test\n    start_time = time.time()\n    model.train()\n    train_loss, train_correct, total_train = 0.0, 0, 0\n    \n    for face, body, labels in train_loader:\n        face, body, labels = face.to(device), body.to(device), labels.to(device)\n        optimizer.zero_grad()\n        logits = model(face, body)\n        loss = criterion(logits, labels)\n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item() * face.size(0)\n        train_correct += (logits.argmax(1) == labels).sum().item()\n        total_train += labels.size(0)\n\n    # Validation\n    model.eval()\n    val_loss, val_correct, total_val = 0.0, 0, 0\n    with torch.no_grad():\n        for face, body, labels in val_loader:\n            face, body, labels = face.to(device), body.to(device), labels.to(device)\n            logits = model(face, body)\n            loss = criterion(logits, labels)\n            val_loss += loss.item() * face.size(0)\n            val_correct += (logits.argmax(1) == labels).sum().item()\n            total_val += labels.size(0)\n\n    print(f\"Epoch {epoch} | Time: {time.time()-start_time:.1f}s | Train Acc: {100*train_correct/total_train:.2f}% | Val Acc: {100*val_correct/total_val:.2f}%\")\n    scheduler.step(val_loss/total_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T18:53:01.056842Z","iopub.execute_input":"2026-04-24T18:53:01.057438Z","iopub.status.idle":"2026-04-24T19:02:11.144304Z","shell.execute_reply.started":"2026-04-24T18:53:01.057406Z","shell.execute_reply":"2026-04-24T19:02:11.143176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport numpy as np\nimport os\nimport json\nimport time\n\nBASE_DIR = \"/kaggle/input/competitions/asl-signs\"\nCSV_PATH = os.path.join(BASE_DIR, \"train.csv\")\nJSON_PATH = os.path.join(BASE_DIR, \"sign_to_prediction_index_map.json\")\n\nwith open(JSON_PATH, 'r') as f:\n    sign_map = json.load(f)\n\nclass GoogleSignDataset(Dataset):\n    def __init__(self, df, base_dir, sign_map, n_frames=30):\n        self.df = df\n        self.base_dir = base_dir\n        self.n_frames = n_frames\n        self.label_map = sign_map\n\n    def __len__(self): \n        return len(self.df)\n\n    def resample(self, x, size):\n        if len(x) >= size: \n            indices = np.linspace(0, len(x) - 1, size).astype(int)\n        else: \n            indices = np.pad(np.arange(len(x)), (0, max(0, size - len(x))), 'edge')\n        return x[indices]\n\n    def load_video(self, path):\n        full_path = os.path.join(self.base_dir, path)\n        try:\n            data = pd.read_parquet(full_path)\n            xyz = data[['x', 'y', 'z']].values.reshape(-1, 543, 3)\n            return np.nan_to_num(xyz, nan=0.0)\n        except: \n            return np.zeros((self.n_frames, 543, 3))\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        landmarks = self.load_video(row['path'])\n        \n        face = landmarks[:, 0:468, :].reshape(len(landmarks), -1)\n        body = landmarks[:, 468:, :].reshape(len(landmarks), -1)\n        \n        face = self.resample(face, self.n_frames)\n        body = self.resample(body, self.n_frames)\n        \n        face = (face - face.mean()) / (face.std() + 1e-6)\n        body = (body - body.mean()) / (body.std() + 1e-6)\n        \n        return torch.tensor(face, dtype=torch.float32), \\\n               torch.tensor(body, dtype=torch.float32), \\\n               torch.tensor(self.label_map[row['sign']], dtype=torch.long)\n\nfull_df = pd.read_csv(CSV_PATH)\ntrain_df, val_df = train_test_split(full_df, test_size=0.10, stratify=full_df['sign'], random_state=42)\nn_classes = len(sign_map)\n\ntrain_loader = DataLoader(GoogleSignDataset(train_df, BASE_DIR, sign_map), batch_size=64, shuffle=True, num_workers=4, pin_memory=True)\nval_loader = DataLoader(GoogleSignDataset(val_df, BASE_DIR, sign_map), batch_size=64, shuffle=False, num_workers=4, pin_memory=True)\n\nclass DualStreamLSTM(nn.Module):\n    def __init__(self, n_classes):\n        super().__init__()\n        self.face_fc = nn.Linear(1404, 256)\n        self.face_lstm = nn.LSTM(256, 256, batch_first=True, num_layers=2, dropout=0.3)\n        \n        self.body_fc = nn.Linear(225, 256)\n        self.body_lstm = nn.LSTM(256, 256, batch_first=True, num_layers=2, dropout=0.3)\n        \n        self.classifier = nn.Sequential(\n            nn.BatchNorm1d(256),\n            nn.Linear(256, 512),\n            nn.ReLU(),\n            nn.Dropout(0.5),\n            nn.Linear(512, n_classes)\n        )\n\n    def forward(self, face, body):\n        f = torch.relu(self.face_fc(face))\n        _, (h_f, _) = self.face_lstm(f)\n        feat_face = h_f[-1] \n        \n        b = torch.relu(self.body_fc(body))\n        _, (h_b, _) = self.body_lstm(b)\n        feat_body = h_b[-1]\n        \n        combined = (feat_face + feat_body) / 2\n        return self.classifier(combined)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = DualStreamLSTM(n_classes).to(device)\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\ncriterion = nn.CrossEntropyLoss()\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3)\n\nbest_val_acc = 0.0\n\nprint(f\"Training started on {device}\")\nfor epoch in range(1, 101):\n    model.train()\n    t_loss, t_acc = 0, 0\n    start_time = time.time()\n    \n    for f, b, l in train_loader:\n        f, b, l = f.to(device), b.to(device), l.to(device)\n        optimizer.zero_grad()\n        out = model(f, b)\n        loss = criterion(out, l)\n        loss.backward()\n        nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n        optimizer.step()\n        t_loss += loss.item()\n        t_acc += (out.argmax(1) == l).sum().item()\n\n    model.eval()\n    v_loss, v_acc = 0, 0\n    with torch.no_grad():\n        for f, b, l in val_loader:\n            f, b, l = f.to(device), b.to(device), l.to(device)\n            out = model(f, b)\n            v_loss += criterion(out, l).item()\n            v_acc += (out.argmax(1) == l).sum().item()\n\n    avg_t_loss = t_loss / len(train_loader)\n    avg_t_acc = t_acc / len(train_df)\n    avg_v_loss = v_loss / len(val_loader)\n    avg_v_acc = v_acc / len(val_df)\n    \n    scheduler.step(avg_v_loss)\n    \n    print(f\"Epoch {epoch:03d} | Time: {time.time()-start_time:.1f}s | Train Acc: {avg_t_acc:.4f} | Val Acc: {avg_v_acc:.4f} | Loss: {avg_v_loss:.4f}\")\n    \n    if avg_v_acc > best_val_acc:\n        best_val_acc = avg_v_acc\n        torch.save(model.state_dict(), \"best_asl_model_avg.pt\")\n        print(f\"New Best Accuracy: {best_val_acc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T19:04:15.976037Z","iopub.execute_input":"2026-04-24T19:04:15.976704Z","iopub.status.idle":"2026-04-25T01:23:33.532844Z","shell.execute_reply.started":"2026-04-24T19:04:15.976668Z","shell.execute_reply":"2026-04-25T01:23:33.525334Z"}},"outputs":[],"execution_count":null}]}