{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport gc\n\n# PyTorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\n\n# Configuration\nDATA_DIR = '/kaggle/input/physionet-ecg-image-digitization'\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, 'train')\nTEST_IMG_DIR = os.path.join(DATA_DIR, 'test')\n\n# Use GPU if available (Crucial for deep learning)\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# Parameters\nIMG_SIZE = 256  # Resize all images to 256x256 for speed\nBATCH_SIZE = 16 # Process 16 images at a time\nEPOCHS = 3      # How many times to study the dataset (Increase this to 10-20 for better scores later)\nLR = 1e-4       # Learning rate (how fast the model changes its mind)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T04:08:06.922045Z","iopub.execute_input":"2026-01-06T04:08:06.922950Z","iopub.status.idle":"2026-01-06T04:08:11.329079Z","shell.execute_reply.started":"2026-01-06T04:08:06.922919Z","shell.execute_reply":"2026-01-06T04:08:11.328270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_image(id_code, split='train'):\n    \"\"\"Loads an image. Returns a blank image if not found.\"\"\"\n    if split == 'train':\n        # Train images are in subfolders. We look for the first .png available.\n        path_folder = os.path.join(TRAIN_IMG_DIR, str(id_code))\n        if os.path.exists(path_folder):\n            files = [f for f in os.listdir(path_folder) if f.endswith('.png')]\n            if files:\n                # Prioritize standard scans if possible, otherwise take the first one\n                # Here we just take the first one for simplicity\n                img_path = os.path.join(path_folder, files[0])\n                try:\n                    img = cv2.imread(img_path)\n                    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n                    return img\n                except:\n                    pass\n    else:\n        # Test images are directly in the test folder\n        img_path = os.path.join(TEST_IMG_DIR, f\"{id_code}.png\")\n        if os.path.exists(img_path):\n            try:\n                img = cv2.imread(img_path)\n                img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n                return img\n            except:\n                pass\n    \n    # Return a black image if file is missing/corrupt\n    return np.zeros((IMG_SIZE, IMG_SIZE, 3), dtype=np.uint8)\n\ndef load_signal(id_code):\n    \"\"\"Loads the ground truth CSV signal for a training ID.\"\"\"\n    path = os.path.join(TRAIN_IMG_DIR, str(id_code), f\"{id_code}.csv\")\n    if os.path.exists(path):\n        return pd.read_csv(path)\n    return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T04:08:11.330578Z","iopub.execute_input":"2026-01-06T04:08:11.331212Z","iopub.status.idle":"2026-01-06T04:08:11.338351Z","shell.execute_reply.started":"2026-01-06T04:08:11.331178Z","shell.execute_reply":"2026-01-06T04:08:11.337613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- REPLACEMENT CELL 3: Dataset Class for Expanded Data ---\nimport cv2\nimport torch\nimport numpy as np\nfrom torch.utils.data import Dataset\nfrom torchvision import transforms\n\nclass ECGDataset(Dataset):\n    def __init__(self, df, transform=None, mode='train'):\n        self.df = df\n        self.transform = transform\n        self.mode = mode\n        # Leads configuration\n        self.leads = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        id_code = row['id']\n        \n        # 1. LOAD IMAGE\n        # If we have the direct path (from our Data Expander step), use it!\n        if 'image_path' in row:\n            img_path = row['image_path']\n        else:\n            # Fallback for Test data (which doesn't have expanded paths)\n            img_path = os.path.join(TEST_IMG_DIR, f\"{id_code}.png\")\n\n        try:\n            img = cv2.imread(img_path)\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        except:\n            # Safety: return black image if file is broken\n            img = np.zeros((256, 256, 3), dtype=np.uint8)\n        \n        # 2. TRANSFORM\n        if self.transform:\n            img = self.transform(img)\n            \n        # 3. HANDLE TARGETS (Training Mode Only)\n        if self.mode == 'train':\n            # Load the signal CSV for this patient\n            sig_df = load_signal(id_code)\n            \n            if sig_df is None:\n                # Return zeros if signal file is missing\n                return img, torch.zeros((12, 1000), dtype=torch.float32)\n\n            signals = []\n            for lead in self.leads:\n                if lead in sig_df.columns:\n                    s = sig_df[lead].values\n                    # Fix NaNs (Empty values)\n                    s = np.nan_to_num(s, nan=0.0)\n                    # Resample to exactly 1000 points\n                    s_resampled = np.interp(np.linspace(0, len(s), 1000), np.arange(len(s)), s)\n                    signals.append(s_resampled)\n                else:\n                    signals.append(np.zeros(1000))\n            \n            return img, torch.tensor(np.array(signals), dtype=torch.float32)\n        \n        return img, id_code\n\n# Standard transforms\ndata_transforms = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((256, 256)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nprint(\"✅ New Dataset Class Defined Successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T04:08:11.339426Z","iopub.execute_input":"2026-01-06T04:08:11.339772Z","iopub.status.idle":"2026-01-06T04:08:11.365129Z","shell.execute_reply.started":"2026-01-06T04:08:11.339737Z","shell.execute_reply":"2026-01-06T04:08:11.364388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ECGResNet(nn.Module):\n    def __init__(self):\n        super(ECGResNet, self).__init__()\n        \n        # 1. Load ResNet18\n        # weights=None is the correct modern way to say \"no internet/pretrained\"\n        self.base_model = models.resnet18(weights=None)\n        \n        # 2. Modify the Final Layer\n        # We wrap Dropout AND Linear together in a Sequence.\n        # This drops internal features (good), not the final output (bad).\n        num_features = self.base_model.fc.in_features\n        \n        self.base_model.fc = nn.Sequential(\n            nn.Dropout(p=0.2),                   # 1. Drop 20% of neuron connections\n            nn.Linear(num_features, 12 * 1000)   # 2. Then predict the 12,000 points\n        )\n        \n    def forward(self, x):\n        batch_size = x.size(0)\n        x = self.base_model(x)\n        # Reshape the flat 12000 vector into (Batch, 12, 1000)\n        return x.view(batch_size, 12, 1000)\n\nmodel = ECGResNet().to(device)\nprint(\"Model initialized successfully (with correct Dropout).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T04:08:11.366393Z","iopub.execute_input":"2026-01-06T04:08:11.366683Z","iopub.status.idle":"2026-01-06T04:08:11.797358Z","shell.execute_reply.started":"2026-01-06T04:08:11.366661Z","shell.execute_reply":"2026-01-06T04:08:11.796739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport glob\nfrom tqdm import tqdm\n\n# Configuration\nDATA_DIR = '/kaggle/input/physionet-ecg-image-digitization'\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, 'train')\n\n# 1. Load the original list of IDs\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, 'train.csv'))\n\n# 2. Create an Empty List to store EVERY image found\nexpanded_data = []\n\nprint(\"Scanning folders for all images... (This might take 1 minute)\")\n\n# 3. Loop through every Patient ID\nfor index, row in tqdm(train_df.iterrows(), total=len(train_df)):\n    id_code = str(row['id'])\n    folder_path = os.path.join(TRAIN_IMG_DIR, id_code)\n    \n    # Check if folder exists\n    if os.path.exists(folder_path):\n        # Find ALL .png files in this folder\n        png_files = [f for f in os.listdir(folder_path) if f.endswith('.png')]\n        \n        # Add each image to our new list\n        for file_name in png_files:\n            full_path = os.path.join(folder_path, file_name)\n            \n            # We save the ID (to find the CSV later) and the direct Image Path\n            expanded_data.append({\n                'id': row['id'],         # Keep the ID to find the target CSV signal\n                'image_path': full_path, # Direct path to this specific image\n                'filename': file_name\n            })\n\n# 4. Create the NEW Huge DataFrame\nexpanded_train_df = pd.DataFrame(expanded_data)\n\nprint(f\"✅ DONE! Original Patient Count: {len(train_df)}\")\nprint(f\"🚀 NEW Expanded Image Count: {len(expanded_train_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T04:08:11.798277Z","iopub.execute_input":"2026-01-06T04:08:11.798540Z","iopub.status.idle":"2026-01-06T04:08:19.885098Z","shell.execute_reply.started":"2026-01-06T04:08:11.798512Z","shell.execute_reply":"2026-01-06T04:08:19.884530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- REPLACEMENT CELL: Start Training on 8,793 Images ---\nimport torch.optim as optim\nfrom tqdm import tqdm\n\n# 1. Config & Split\n# We use the NEW expanded dataframe\nval_size = int(len(expanded_train_df) * 0.1)\ntrain_subset = expanded_train_df.iloc[:-val_size]\nval_subset = expanded_train_df.iloc[-val_size:]\n\nprint(f\"🔥 Training on {len(train_subset)} images\")\nprint(f\"🧪 Validating on {len(val_subset)} images\")\n\n# 2. Setup Loaders\n# Note: Ensure you ran the 'ECGDataset' class cell from my previous message!\ntrain_ds = ECGDataset(train_subset, transform=data_transforms, mode='train')\nval_ds = ECGDataset(val_subset, transform=data_transforms, mode='train')\n\ntrain_loader = DataLoader(train_ds, batch_size=32, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_ds, batch_size=32, shuffle=False, num_workers=2)\n\n# 3. Model Setup\nmodel = ECGResNet().to(device)\n\n# Optional: Load offline weights if available (Not mandatory, but helps)\nweight_path = \"/kaggle/input/resnet18/resnet18.pth\"\nif os.path.exists(weight_path):\n    try:\n        model.base_model.load_state_dict(torch.load(weight_path))\n        print(\"✅ Loaded Offline ResNet18 Weights\")\n    except:\n        print(\"⚠️ Weights found but load failed. Training from scratch.\")\nelse:\n    print(\"ℹ️ No offline weights found. Training from scratch (This is fine now that we have 8k images!)\")\n\n# 4. Optimizer & Scheduler\ncriterion = nn.L1Loss() # Mean Absolute Error\noptimizer = optim.AdamW(model.parameters(), lr=1e-3, weight_decay=1e-5)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=2)\n\n# 5. The Training Loop\nEPOCHS = 8 # 8 Epochs is enough for this amount of data\nbest_val_loss = float('inf')\n\nprint(\"\\n🚀 STARTING MASSIVE TRAINING...\")\n\nfor epoch in range(EPOCHS):\n    model.train()\n    running_loss = 0.0\n    loop = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{EPOCHS}\")\n    \n    for images, targets in loop:\n        images = images.to(device)\n        targets = targets.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, targets)\n        \n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n        loop.set_postfix(loss=loss.item())\n        \n    avg_train_loss = running_loss / len(train_loader)\n    \n    # Validation\n    model.eval()\n    val_loss = 0.0\n    with torch.no_grad():\n        for images, targets in val_loader:\n            images = images.to(device)\n            targets = targets.to(device)\n            outputs = model(images)\n            loss = criterion(outputs, targets)\n            val_loss += loss.item()\n    \n    avg_val_loss = val_loss / len(val_loader)\n    \n    # Scheduler\n    scheduler.step(avg_val_loss)\n    \n    print(f\"Epoch {epoch+1} Result: Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_val_loss:.4f}\")\n    \n    # Save Best Model\n    if avg_val_loss < best_val_loss:\n        best_val_loss = avg_val_loss\n        torch.save(model.state_dict(), \"best_model.pth\")\n        print(\">>> 💾 Model Improved! Saved 'best_model.pth'\")\n\nprint(\"✅ Training Complete! You are ready to generate the submission.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T04:08:19.885947Z","iopub.execute_input":"2026-01-06T04:08:19.886264Z","iopub.status.idle":"2026-01-06T07:23:40.007916Z","shell.execute_reply.started":"2026-01-06T04:08:19.886240Z","shell.execute_reply":"2026-01-06T07:23:40.003615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- FINAL CELL: Generate Submission ---\nimport pandas as pd\nimport numpy as np\nimport torch\nimport os\nfrom tqdm import tqdm\nfrom torch.utils.data import DataLoader\n\n# 1. Load the Best Model\nmodel = ECGResNet().to(device)\n# Load the weights we just saved in the training step\nmodel.load_state_dict(torch.load(\"best_model.pth\"))\nmodel.eval()\n\n# 2. Prepare Test Data\n# Load the test requirements\ntest_df = pd.read_csv(os.path.join(DATA_DIR, 'test.csv'))\nunique_test_ids = test_df['id'].unique()\n\n# Create Dataset in TEST mode\ntest_ds = ECGDataset(pd.DataFrame({'id': unique_test_ids}), transform=data_transforms, mode='test')\ntest_loader = DataLoader(test_ds, batch_size=32, shuffle=False, num_workers=2)\n\nprint(f\"📝 Generating predictions for {len(unique_test_ids)} patients...\")\n\n# 3. Prediction Loop\ncsv_file = 'submission.csv'\n# Write header\nwith open(csv_file, 'w') as f:\n    f.write('id,value\\n')\n\n# Map lead names to indices (Standard 12-lead order)\nlead_map = {'I':0, 'II':1, 'III':2, 'aVR':3, 'aVL':4, 'aVF':5, \n            'V1':6, 'V2':7, 'V3':8, 'V4':9, 'V5':10, 'V6':11}\n\nprint(\"Running inference...\")\n\nwith torch.no_grad():\n    for images, id_codes in tqdm(test_loader):\n        images = images.to(device)\n        \n        # Get Model Predictions (Shape: Batch x 12 x 1000)\n        preds = model(images).cpu().numpy()\n        \n        # temporary lists to store data for this batch\n        batch_ids = []\n        batch_values = []\n        \n        # Loop through each patient in this batch\n        for i, current_id in enumerate(id_codes):\n            # Fix ID type if it's a tensor (converts tensor(123) -> 123)\n            if isinstance(current_id, torch.Tensor):\n                current_id = current_id.item()\n            \n            # Find what the submission file wants for THIS patient\n            reqs = test_df[test_df['id'] == current_id]\n            \n            for _, row in reqs.iterrows():\n                lead_name = row['lead']\n                required_rows = int(row['number_of_rows'])\n                lead_idx = lead_map.get(lead_name, 0)\n                \n                # Extract the 1000 points for this specific lead\n                raw_signal = preds[i, lead_idx]\n                \n                # If the submission asks for a different length, resample it\n                if required_rows > 0:\n                    resampled = np.interp(\n                        np.linspace(0, 1000, required_rows),\n                        np.linspace(0, 1000, 1000),\n                        raw_signal\n                    )\n                    \n                    # Create the formatted IDs: \"PatientID_Index_Lead\"\n                    ids = [f\"{current_id}_{k}_{lead_name}\" for k in range(len(resampled))]\n                    \n                    batch_ids.extend(ids)\n                    batch_values.extend(resampled)\n\n        # Write this batch to CSV immediately (Saves RAM)\n        if batch_ids:\n            temp_df = pd.DataFrame({'id': batch_ids, 'value': batch_values})\n            temp_df.to_csv(csv_file, mode='a', header=False, index=False)\n\nprint(\"✅ SUCCESS! 'submission.csv' created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T07:42:04.671458Z","iopub.execute_input":"2026-01-06T07:42:04.672318Z","iopub.status.idle":"2026-01-06T07:42:06.380813Z","shell.execute_reply.started":"2026-01-06T07:42:04.672285Z","shell.execute_reply":"2026-01-06T07:42:06.379981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 1. Read the file you just created\ndf_sub = pd.read_csv('submission.csv')\n\n# 2. Print the top 10 rows\nprint(\"--- 📄 Submission File Preview (Top 10) ---\")\nprint(df_sub.head(10))\n\n# 3. Print the bottom 5 rows\nprint(\"\\n--- 📄 Tail Preview (Bottom 5) ---\")\nprint(df_sub.tail(5))\n\n# 4. Final Stats\nprint(f\"\\n✅ Total Rows Generated: {len(df_sub)}\")\nprint(f\"❌ NaN/Empty Values: {df_sub.isna().sum().sum()} (Should be 0)\")\n\n# 5. Quick Sanity Check\n# Ensure values are not all exactly 0.0 (which would mean model failed)\nnon_zero_count = (df_sub['value'] != 0).sum()\nprint(f\"⚡ Non-Zero Predictions: {non_zero_count} (Should be high!)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-06T07:44:25.104252Z","iopub.execute_input":"2026-01-06T07:44:25.105082Z","iopub.status.idle":"2026-01-06T07:44:25.176831Z","shell.execute_reply.started":"2026-01-06T07:44:25.105046Z","shell.execute_reply":"2026-01-06T07:44:25.176194Z"}},"outputs":[],"execution_count":null}]}