{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":14499569,"sourceType":"datasetVersion","datasetId":9261206},{"sourceId":14508463,"sourceType":"datasetVersion","datasetId":9266551}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PhysioNet ECG Digitization: Inference (Final Robust)\n\nThis version fixes **Submission Scoring Errors** by:\n1. **Deduplicating IDs**: Ensures we process each image only once, even if `test.csv` has multiple rows.\n2. **Zero-Filling**: If an image fails to load/predict, we output Zeros instead of skipping. This ensures complete submission.\n3. **Zero-Dependency**: Uses PyTorch + Standard Lib only.","metadata":{}},{"cell_type":"code","source":"# Cell 1: Imports and path sanity check (NO INTERNET)\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom PIL import Image\n\nprint(\"Torch version:\", torch.__version__)\nprint(\"CUDA available:\", torch.cuda.is_available())\n\n# ---- UPDATE THESE PATHS IF NEEDED ----\nDATA_ROOT = \"/kaggle/input/physionet-ecg-image-digitization\"\nWEIGHTS_ROOT = \"/kaggle/input\"  # weights dataset should be inside here\n\n# Check main competition files\nprint(\"\\nChecking competition data files:\")\nprint(\"train.csv exists:\", os.path.exists(os.path.join(DATA_ROOT, \"train.csv\")))\nprint(\"test.csv exists:\", os.path.exists(os.path.join(DATA_ROOT, \"test.csv\")))\n\n# Load test.csv (lightweight check)\ntest_csv_path = os.path.join(DATA_ROOT, \"test.csv\")\ntest_df = pd.read_csv(test_csv_path)\nprint(\"\\nLoaded test.csv\")\nprint(\"Number of rows in test.csv:\", len(test_df))\nprint(\"Columns:\", test_df.columns.tolist())\n\n# List datasets available under /kaggle/input\nprint(\"\\nAvailable datasets under /kaggle/input:\")\nfor d in os.listdir(\"/kaggle/input\"):\n    print(\" -\", d)\n\n# Try to locate .pth files\nprint(\"\\nSearching for .pth files:\")\npth_files = []\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    for f in files:\n        if f.endswith(\".pth\"):\n            pth_files.append(os.path.join(root, f))\n\nprint(\"Found\", len(pth_files), \".pth files\")\nfor p in pth_files:\n    print(\" \", p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:00:51.140109Z","iopub.execute_input":"2026-01-16T06:00:51.140845Z","iopub.status.idle":"2026-01-16T06:01:01.378487Z","shell.execute_reply.started":"2026-01-16T06:00:51.140817Z","shell.execute_reply":"2026-01-16T06:01:01.377818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 1: Inspect checkpoint structure and tensor shapes (SOURCE OF TRUTH)\n\nimport torch\nfrom collections import defaultdict\n\nCKPT_PATH = \"/kaggle/input/ecg-model-1/ecg_model_e1.pth\"\n\nckpt = torch.load(CKPT_PATH, map_location=\"cpu\")\n\n# unwrap common wrappers\nif isinstance(ckpt, dict) and \"state_dict\" in ckpt:\n    state = ckpt[\"state_dict\"]\nelif isinstance(ckpt, dict) and \"model\" in ckpt:\n    state = ckpt[\"model\"]\nelse:\n    state = ckpt\n\nprint(\"Total parameters:\", len(state))\n\n# group keys\ngroups = defaultdict(list)\nfor k in state.keys():\n    prefix = k.split(\".\")[0]\n    groups[prefix].append(k)\n\nprint(\"\\nTop-level modules:\")\nfor g in groups:\n    print(\" -\", g)\n\n# ---- Encoder inspection ----\nprint(\"\\n[ENCODER SAMPLE KEYS]\")\nfor k in list(groups[\"encoder\"])[:10]:\n    print(k, state[k].shape)\n\n# ---- Decoder inspection ----\ndecoder_blocks = defaultdict(list)\nfor k in groups[\"decoder\"]:\n    parts = k.split(\".\")\n    if len(parts) > 2 and parts[1] == \"blocks\":\n        block_id = parts[2]\n        decoder_blocks[block_id].append(k)\n\nprint(\"\\n[DECODER BLOCKS FOUND]\")\nfor b in sorted(decoder_blocks.keys()):\n    print(f\"\\nBlock {b}:\")\n    for k in decoder_blocks[b]:\n        print(\" \", k, state[k].shape)\n\n# ---- Segmentation head ----\nprint(\"\\n[SEGMENTATION HEAD]\")\nfor k in groups[\"segmentation_head\"]:\n    print(k, state[k].shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:44:58.117822Z","iopub.execute_input":"2026-01-16T06:44:58.118414Z","iopub.status.idle":"2026-01-16T06:45:02.508361Z","shell.execute_reply.started":"2026-01-16T06:44:58.118383Z","shell.execute_reply":"2026-01-16T06:45:02.507661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 2: Exact UNet-style model reconstructed from checkpoint\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport timm\n\n# ----- basic conv block -----\nclass ConvBNReLU(nn.Sequential):\n    def __init__(self, in_ch, out_ch):\n        super().__init__(\n            nn.Conv2d(in_ch, out_ch, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True)\n        )\n\n# ----- decoder block -----\nclass DecoderBlock(nn.Module):\n    def __init__(self, in_ch, out_ch):\n        super().__init__()\n        self.conv1 = ConvBNReLU(in_ch, out_ch)\n        self.conv2 = ConvBNReLU(out_ch, out_ch)\n\n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        return x\n\n# ----- full model -----\n# Cell 3: Fix forward() for timm ResNet (act1 instead of relu)\n\n# Cell 5: Final fix — remove encoder.fc (num_classes=0)\n\nclass ECGUNet(nn.Module):\n    def __init__(self):\n        super().__init__()\n\n        # IMPORTANT FIX: num_classes=0 removes fc layer\n        self.encoder = timm.create_model(\n            \"resnet34\",\n            pretrained=False,\n            num_classes=0\n        )\n\n        self.decoder = nn.ModuleDict({\n            \"blocks\": nn.ModuleList([\n                DecoderBlock(768, 256),\n                DecoderBlock(384, 128),\n                DecoderBlock(192, 64),\n                DecoderBlock(128, 32),\n                DecoderBlock(32, 16),\n            ])\n        })\n\n        self.segmentation_head = nn.Sequential(\n            nn.Conv2d(16, 1, kernel_size=3, padding=1)\n        )\n\n    def forward(self, x):\n        x0 = self.encoder.conv1(x)\n        x0 = self.encoder.bn1(x0)\n        x0 = self.encoder.act1(x0)\n        x1 = self.encoder.maxpool(x0)\n\n        x1 = self.encoder.layer1(x1)\n        x2 = self.encoder.layer2(x1)\n        x3 = self.encoder.layer3(x2)\n        x4 = self.encoder.layer4(x3)\n\n        d0 = F.interpolate(x4, scale_factor=2, mode=\"bilinear\", align_corners=False)\n        d0 = torch.cat([d0, x3], dim=1)\n        d0 = self.decoder[\"blocks\"][0](d0)\n\n        d1 = F.interpolate(d0, scale_factor=2, mode=\"bilinear\", align_corners=False)\n        d1 = torch.cat([d1, x2], dim=1)\n        d1 = self.decoder[\"blocks\"][1](d1)\n\n        d2 = F.interpolate(d1, scale_factor=2, mode=\"bilinear\", align_corners=False)\n        d2 = torch.cat([d2, x1], dim=1)\n        d2 = self.decoder[\"blocks\"][2](d2)\n\n        d3 = F.interpolate(d2, scale_factor=2, mode=\"bilinear\", align_corners=False)\n        d3 = torch.cat([d3, x0], dim=1)\n        d3 = self.decoder[\"blocks\"][3](d3)\n\n        d4 = F.interpolate(d3, scale_factor=2, mode=\"bilinear\", align_corners=False)\n        d4 = self.decoder[\"blocks\"][4](d4)\n\n        return self.segmentation_head(d4)\n\n\n# ---- sanity check ----\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = ECGUNet().to(device)\n\ndummy = torch.randn(1, 3, 224, 224).to(device)\nwith torch.no_grad():\n    y = model(dummy)\n\nprint(\"Model rebuilt (no FC)\")\nprint(\"Output shape:\", y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:50:13.399882Z","iopub.execute_input":"2026-01-16T06:50:13.400505Z","iopub.status.idle":"2026-01-16T06:50:13.744539Z","shell.execute_reply.started":"2026-01-16T06:50:13.400472Z","shell.execute_reply":"2026-01-16T06:50:13.743931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 6: FINAL strict load check\n\nWEIGHT_PATH = \"/kaggle/input/ecg-model-1/ecg_model_e1.pth\"\n\nstate = torch.load(WEIGHT_PATH, map_location=device)\n\n# unwrap if needed\nif isinstance(state, dict) and \"state_dict\" in state:\n    state = state[\"state_dict\"]\nelif isinstance(state, dict) and \"model\" in state:\n    state = state[\"model\"]\n\nprint(\"Attempting FINAL strict load...\")\n\nmodel.load_state_dict(state, strict=True)\n\nprint(\"✅ STRICT LOAD PASSED\")\n\n# final forward sanity\nmodel.eval()\ndummy = torch.randn(1, 3, 224, 224).to(device)\nwith torch.no_grad():\n    y = model(dummy)\n\nprint(\"Post-load forward OK\")\nprint(\"Output shape:\", y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:50:56.268708Z","iopub.execute_input":"2026-01-16T06:50:56.269356Z","iopub.status.idle":"2026-01-16T06:50:56.418105Z","shell.execute_reply.started":"2026-01-16T06:50:56.269327Z","shell.execute_reply":"2026-01-16T06:50:56.417406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 7: Load all 5 trained models for ensemble inference\n\nWEIGHT_PATHS = [\n    \"/kaggle/input/ecg-model-1/ecg_model_e1.pth\",\n    \"/kaggle/input/ecg-model-1/ecg_model_e2.pth\",\n    \"/kaggle/input/ecg-model-1/ecg_model_e3.pth\",\n    \"/kaggle/input/ecg-model-1/ecg_model_e4.pth\",\n    \"/kaggle/input/ecg-model-1/ecg_model_e5.pth\",\n]\n\nmodels = []\n\nfor i, path in enumerate(WEIGHT_PATHS):\n    m = ECGUNet().to(device)\n    \n    state = torch.load(path, map_location=device)\n    if isinstance(state, dict) and \"state_dict\" in state:\n        state = state[\"state_dict\"]\n    elif isinstance(state, dict) and \"model\" in state:\n        state = state[\"model\"]\n\n    m.load_state_dict(state, strict=True)\n    m.eval()\n    models.append(m)\n\n    print(f\"Model {i+1} loaded successfully\")\n\nprint(f\"\\nTotal models loaded: {len(models)}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:51:30.070832Z","iopub.execute_input":"2026-01-16T06:51:30.071183Z","iopub.status.idle":"2026-01-16T06:51:35.763752Z","shell.execute_reply.started":"2026-01-16T06:51:30.071128Z","shell.execute_reply":"2026-01-16T06:51:35.763088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 8: Single test image preprocessing + ensemble inference sanity check\nimport os\nimport cv2\nimport numpy as np\n\n# ---- basic preprocessing (same as training assumptions) ----\ndef preprocess_image(img_path, img_size=224):\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (img_size, img_size))\n    img = img.astype(np.float32) / 255.0\n    img = img.transpose(2, 0, 1)  # HWC -> CHW\n    return torch.tensor(img).unsqueeze(0)\n\n# Pick ONE test image\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\ntest_images = sorted(os.listdir(TEST_IMG_DIR))\n\nprint(\"Number of test images:\", len(test_images))\nprint(\"Using test image:\", test_images[0])\n\nimg_path = os.path.join(TEST_IMG_DIR, test_images[0])\nx = preprocess_image(img_path).to(device)\n\n# ---- ensemble inference ----\nwith torch.no_grad():\n    preds = []\n    for m in models:\n        y = m(x)\n        preds.append(y)\n    pred = torch.mean(torch.stack(preds), dim=0)\n\nprint(\"Ensemble output shape:\", pred.shape)\nprint(\"Min value:\", pred.min().item())\nprint(\"Max value:\", pred.max().item())\nprint(\"Mean value:\", pred.mean().item())\nprint(\"Any NaNs:\", torch.isnan(pred).any().item())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:53:01.645971Z","iopub.execute_input":"2026-01-16T06:53:01.646735Z","iopub.status.idle":"2026-01-16T06:53:02.120897Z","shell.execute_reply.started":"2026-01-16T06:53:01.646705Z","shell.execute_reply":"2026-01-16T06:53:02.120122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 9: Convert segmentation map to 1D ECG waveform (sanity)\n\ndef segmentation_to_waveform(seg_map):\n    \"\"\"\n    seg_map: torch.Tensor of shape (1, 1, H, W)\n    returns: 1D numpy array of length W\n    \"\"\"\n    seg_map = seg_map.squeeze(0).squeeze(0)  # (H, W)\n\n    # Convert to numpy\n    seg_np = seg_map.cpu().numpy()\n\n    # Aggregate vertically (mean works well; median is also possible)\n    waveform = seg_np.mean(axis=0)  # (W,)\n\n    return waveform\n\n\nwaveform = segmentation_to_waveform(pred)\n\nprint(\"Waveform length:\", waveform.shape[0])\nprint(\"Min:\", waveform.min())\nprint(\"Max:\", waveform.max())\nprint(\"Mean:\", waveform.mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:54:26.979906Z","iopub.execute_input":"2026-01-16T06:54:26.980477Z","iopub.status.idle":"2026-01-16T06:54:26.986982Z","shell.execute_reply.started":"2026-01-16T06:54:26.980448Z","shell.execute_reply":"2026-01-16T06:54:26.986386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 10: Resample 1D waveform to base length = 2000\n\ndef resample_waveform(waveform, target_len=2000):\n    \"\"\"\n    waveform: 1D numpy array\n    returns: 1D numpy array of length target_len\n    \"\"\"\n    x_old = np.linspace(0, 1, len(waveform))\n    x_new = np.linspace(0, 1, target_len)\n    return np.interp(x_new, x_old, waveform)\n\n\nwaveform_2000 = resample_waveform(waveform, target_len=2000)\n\nprint(\"Resampled waveform length:\", len(waveform_2000))\nprint(\"Min:\", waveform_2000.min())\nprint(\"Max:\", waveform_2000.max())\nprint(\"Mean:\", waveform_2000.mean())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:55:00.560418Z","iopub.execute_input":"2026-01-16T06:55:00.561082Z","iopub.status.idle":"2026-01-16T06:55:00.567266Z","shell.execute_reply.started":"2026-01-16T06:55:00.561054Z","shell.execute_reply":"2026-01-16T06:55:00.566478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11: Convert base waveform to lead-specific waveform\n\ndef build_lead_waveform(\n    base_waveform,\n    fs,\n    lead,\n    number_of_rows\n):\n    \"\"\"\n    base_waveform: numpy array of length 2000 (10s canonical)\n    fs: sampling frequency from test.csv\n    lead: ECG lead name (string)\n    number_of_rows: expected output length for this lead\n    \"\"\"\n\n    # Duration logic\n    if lead == \"II\":\n        duration_sec = 10.0\n        wf = base_waveform\n    else:\n        duration_sec = 2.5\n        wf = base_waveform[:int(2000 * (duration_sec / 10.0))]\n\n    # Resample to match required rows\n    x_old = np.linspace(0, duration_sec, len(wf))\n    x_new = np.linspace(0, duration_sec, number_of_rows)\n    lead_waveform = np.interp(x_new, x_old, wf)\n\n    return lead_waveform\n\n\n# ---- sanity test ----\ntest_fs = 500\ntest_lead = \"V1\"\ntest_rows = int(test_fs * 2.5)\n\nlw = build_lead_waveform(\n    waveform_2000,\n    fs=test_fs,\n    lead=test_lead,\n    number_of_rows=test_rows\n)\n\nprint(\"Lead:\", test_lead)\nprint(\"Expected rows:\", test_rows)\nprint(\"Actual rows:\", len(lw))\nprint(\"Min:\", lw.min())\nprint(\"Max:\", lw.max())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:55:37.149429Z","iopub.execute_input":"2026-01-16T06:55:37.149954Z","iopub.status.idle":"2026-01-16T06:55:37.156885Z","shell.execute_reply.started":"2026-01-16T06:55:37.149924Z","shell.execute_reply":"2026-01-16T06:55:37.156052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 12: Full inference pipeline for ONE test ECG (sanity check)\n\nimport pandas as pd\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\n# Pick ONE base ECG id\nbase_id = test_df.iloc[0][\"id\"]\nprint(\"Using base_id:\", base_id)\n\nimg_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\nx = preprocess_image(img_path).to(device)\n\n# Ensemble inference\nwith torch.no_grad():\n    preds = []\n    for m in models:\n        preds.append(m(x))\n    pred = torch.mean(torch.stack(preds), dim=0)\n\n# Build base waveform\nbase_waveform = segmentation_to_waveform(pred)\nbase_waveform = resample_waveform(base_waveform, target_len=2000)\n\n# Process all leads for this base_id\nsubset = test_df[test_df[\"id\"] == base_id]\n\nprint(\"\\nGenerated leads:\")\nfor _, row in subset.iterrows():\n    lead = row[\"lead\"]\n    fs = row[\"fs\"]\n    n_rows = row[\"number_of_rows\"]\n\n    lw = build_lead_waveform(\n        base_waveform,\n        fs=fs,\n        lead=lead,\n        number_of_rows=n_rows\n    )\n\n    print(\n        f\"Lead {lead}: expected {n_rows}, got {len(lw)} | \"\n        f\"min={lw.min():.3f}, max={lw.max():.3f}\"\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:56:16.519131Z","iopub.execute_input":"2026-01-16T06:56:16.519485Z","iopub.status.idle":"2026-01-16T06:56:16.927418Z","shell.execute_reply.started":"2026-01-16T06:56:16.519459Z","shell.execute_reply":"2026-01-16T06:56:16.926641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 13: Generate full submission.csv (FINAL)\n\nimport pandas as pd\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\nsubmission_rows = []\n\nunique_ids = test_df[\"id\"].unique()\nprint(\"Total ECGs to process:\", len(unique_ids))\n\nfor base_id in tqdm(unique_ids):\n    # Load image\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    # Ensemble inference\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    # Base waveform\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    # All leads for this ECG\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lead_waveform = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        # Build submission rows\n        for i, val in enumerate(lead_waveform):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\n# Create DataFrame\nsubmission_df = pd.DataFrame(submission_rows)\n\n# Save submission\noutput_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(output_path, index=False)\n\nprint(\"\\nSubmission saved to:\", output_path)\nprint(\"Total rows:\", len(submission_df))\nsubmission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T06:57:30.241341Z","iopub.execute_input":"2026-01-16T06:57:30.241858Z","iopub.status.idle":"2026-01-16T06:57:30.771617Z","shell.execute_reply.started":"2026-01-16T06:57:30.241823Z","shell.execute_reply":"2026-01-16T06:57:30.770996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 14: Regenerate submission with normalization + polarity fix (run only)\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef normalize_and_fix_polarity(w):\n    w = w - np.mean(w)\n    std = np.std(w)\n    if std > 1e-6:\n        w = w / std\n    # polarity fix: make dominant peak positive\n    if np.abs(w.min()) > np.abs(w.max()):\n        w = -w\n    return w\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = normalize_and_fix_polarity(lw)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:27:25.650483Z","iopub.execute_input":"2026-01-16T07:27:25.650791Z","iopub.status.idle":"2026-01-16T07:27:26.150264Z","shell.execute_reply.started":"2026-01-16T07:27:25.650765Z","shell.execute_reply":"2026-01-16T07:27:26.149458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 15: Add light temporal smoothing + derivative sharpening (run only)\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef smooth_and_sharpen(w, k=7, alpha=0.3):\n    # moving average smoothing\n    kernel = np.ones(k) / k\n    w_s = np.convolve(w, kernel, mode=\"same\")\n    # derivative sharpening\n    dw = np.gradient(w_s)\n    w_out = w_s + alpha * dw\n    # re-normalize\n    w_out = w_out - np.mean(w_out)\n    std = np.std(w_out)\n    if std > 1e-6:\n        w_out = w_out / std\n    # polarity fix\n    if np.abs(w_out.min()) > np.abs(w_out.max()):\n        w_out = -w_out\n    return w_out\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = smooth_and_sharpen(lw)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:27:57.019994Z","iopub.execute_input":"2026-01-16T07:27:57.020549Z","iopub.status.idle":"2026-01-16T07:27:57.534340Z","shell.execute_reply.started":"2026-01-16T07:27:57.020521Z","shell.execute_reply":"2026-01-16T07:27:57.533673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 16: Bandpass-style cleanup (detrend + lowpass) + robust scaling (run only)\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef bandpass_like(w, low_k=101, high_k=7):\n    # high-pass via detrending with large moving average\n    low_kernel = np.ones(low_k) / low_k\n    trend = np.convolve(w, low_kernel, mode=\"same\")\n    hp = w - trend\n    # low-pass via small moving average\n    high_kernel = np.ones(high_k) / high_k\n    lp = np.convolve(hp, high_kernel, mode=\"same\")\n    # robust scale (MAD)\n    med = np.median(lp)\n    mad = np.median(np.abs(lp - med)) + 1e-6\n    out = (lp - med) / mad\n    # polarity fix\n    if np.abs(out.min()) > np.abs(out.max()):\n        out = -out\n    return out\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = bandpass_like(lw)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:28:29.519818Z","iopub.execute_input":"2026-01-16T07:28:29.520497Z","iopub.status.idle":"2026-01-16T07:28:30.043312Z","shell.execute_reply.started":"2026-01-16T07:28:29.520456Z","shell.execute_reply":"2026-01-16T07:28:30.042698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 17: SAFE amplitude control (hard clamp + unit-variance rescale) — run only\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef safe_scale(w, clip_z=5.0):\n    # center\n    w = w - np.mean(w)\n    # unit variance\n    std = np.std(w)\n    if std > 1e-6:\n        w = w / std\n    # hard clamp to avoid explosions\n    w = np.clip(w, -clip_z, clip_z)\n    # re-center after clip\n    w = w - np.mean(w)\n    return w\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = safe_scale(lw)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:28:56.780525Z","iopub.execute_input":"2026-01-16T07:28:56.781055Z","iopub.status.idle":"2026-01-16T07:28:57.300662Z","shell.execute_reply.started":"2026-01-16T07:28:56.781029Z","shell.execute_reply":"2026-01-16T07:28:57.300062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 18: FIX constant-output bug — minimal processing ONLY (run only)\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef minimal_normalize(w):\n    # center\n    w = w - np.mean(w)\n    # scale (no clipping, no MAD, no derivative)\n    std = np.std(w)\n    if std > 1e-6:\n        w = w / std\n    # polarity fix only\n    if np.abs(w.min()) > np.abs(w.max()):\n        w = -w\n    return w\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = minimal_normalize(lw)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:29:25.183097Z","iopub.execute_input":"2026-01-16T07:29:25.183844Z","iopub.status.idle":"2026-01-16T07:29:25.699607Z","shell.execute_reply.started":"2026-01-16T07:29:25.183815Z","shell.execute_reply":"2026-01-16T07:29:25.698971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FINAL CELL: Stable submission (baseline-safe, no aggressive post-processing)\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport os\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef stable_normalize(w):\n    # remove mean\n    w = w - np.mean(w)\n    # scale to unit variance (only once)\n    std = np.std(w)\n    if std > 1e-6:\n        w = w / std\n    return w\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = stable_normalize(lw)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:30:00.101003Z","iopub.execute_input":"2026-01-16T07:30:00.101669Z","iopub.status.idle":"2026-01-16T07:30:00.619090Z","shell.execute_reply.started":"2026-01-16T07:30:00.101640Z","shell.execute_reply":"2026-01-16T07:30:00.618436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FINAL TEST CELL: global polarity flip (only change is sign inversion)\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport os\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef stable_normalize(w):\n    w = w - np.mean(w)\n    std = np.std(w)\n    if std > 1e-6:\n        w = w / std\n    return w\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = -stable_normalize(lw)  # ONLY difference vs previous run\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:30:29.984727Z","iopub.execute_input":"2026-01-16T07:30:29.985315Z","iopub.status.idle":"2026-01-16T07:30:30.507182Z","shell.execute_reply.started":"2026-01-16T07:30:29.985288Z","shell.execute_reply":"2026-01-16T07:30:30.506554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# NEXT CELL (last attempt that still respects the metric): per-lead energy match + zero-mean\n# Rationale: SNR penalizes shape mismatch more than scale, but extreme scale hurts correlation.\n# This keeps shape, fixes DC, and normalizes RMS per lead without collapsing variance.\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport os\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef rms_normalize(w, target_rms=1.0):\n    w = w - np.mean(w)\n    rms = np.sqrt(np.mean(w**2)) + 1e-8\n    w = w * (target_rms / rms)\n    return w\n\nsubmission_rows = []\nunique_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(unique_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    x = preprocess_image(img_path).to(device)\n\n    with torch.no_grad():\n        preds = []\n        for m in models:\n            preds.append(m(x))\n        pred = torch.mean(torch.stack(preds), dim=0)\n\n    base_waveform = segmentation_to_waveform(pred)\n    base_waveform = resample_waveform(base_waveform, target_len=2000)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        lw = build_lead_waveform(\n            base_waveform,\n            fs=fs,\n            lead=lead,\n            number_of_rows=n_rows\n        )\n\n        lw = rms_normalize(lw, target_rms=1.0)\n\n        for i, val in enumerate(lw):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nout_path, len(submission_df), submission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:30:59.861397Z","iopub.execute_input":"2026-01-16T07:30:59.861976Z","iopub.status.idle":"2026-01-16T07:31:00.377576Z","shell.execute_reply.started":"2026-01-16T07:30:59.861952Z","shell.execute_reply":"2026-01-16T07:31:00.376854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 1: Split ECG image into 12 lead regions (no model change yet)\n\nimport cv2\nimport numpy as np\nimport os\n\ndef split_ecg_leads(img_path):\n    \"\"\"\n    Assumes standard ECG layout:\n    3 rows × 4 columns = 12 leads\n    \"\"\"\n    img = cv2.imread(img_path, cv2.IMREAD_COLOR)\n    h, w, _ = img.shape\n\n    lead_h = h // 3\n    lead_w = w // 4\n\n    leads = {}\n\n    lead_names = [\n        [\"I\", \"II\", \"III\", \"aVR\"],\n        [\"aVL\", \"aVF\", \"V1\", \"V2\"],\n        [\"V3\", \"V4\", \"V5\", \"V6\"]\n    ]\n\n    for r in range(3):\n        for c in range(4):\n            y1 = r * lead_h\n            y2 = (r + 1) * lead_h\n            x1 = c * lead_w\n            x2 = (c + 1) * lead_w\n\n            lead_img = img[y1:y2, x1:x2]\n            lead_name = lead_names[r][c]\n            leads[lead_name] = lead_img\n\n    return leads\n\n\n# --- sanity check on one test image ---\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\nsample_img = os.listdir(TEST_IMG_DIR)[0]\nleads = split_ecg_leads(os.path.join(TEST_IMG_DIR, sample_img))\n\nprint(\"Extracted leads:\", list(leads.keys()))\nfor k, v in leads.items():\n    print(k, v.shape)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:39:56.759848Z","iopub.execute_input":"2026-01-16T07:39:56.760358Z","iopub.status.idle":"2026-01-16T07:39:56.842513Z","shell.execute_reply.started":"2026-01-16T07:39:56.760330Z","shell.execute_reply":"2026-01-16T07:39:56.841659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 2: Run model per lead crop and extract 12 separate waveforms\n\nimport torch\nimport cv2\nimport numpy as np\n\ndef preprocess_lead_img(lead_img):\n    # reuse your existing preprocessing logic, adapted for array input\n    img = cv2.resize(lead_img, (224, 224))\n    img = img.astype(np.float32) / 255.0\n    img = torch.from_numpy(img).permute(2, 0, 1).unsqueeze(0)\n    return img.to(device)\n\ndef infer_lead_waveforms(img_path):\n    leads = split_ecg_leads(img_path)\n    lead_waveforms = {}\n\n    for lead_name, lead_img in leads.items():\n        x = preprocess_lead_img(lead_img)\n\n        with torch.no_grad():\n            preds = []\n            for m in models:\n                preds.append(m(x))\n            pred = torch.mean(torch.stack(preds), dim=0)\n\n        # mask → waveform (your existing function)\n        w = segmentation_to_waveform(pred)\n        w = resample_waveform(w, target_len=2000)\n\n        lead_waveforms[lead_name] = w\n\n    return lead_waveforms\n\n\n# ---- sanity check on one image ----\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\nsample_img = os.listdir(TEST_IMG_DIR)[0]\n\nlead_waves = infer_lead_waveforms(os.path.join(TEST_IMG_DIR, sample_img))\n\nprint(\"Leads inferred:\", list(lead_waves.keys()))\nfor k, v in lead_waves.items():\n    print(k, len(v), f\"min={v.min():.3f}\", f\"max={v.max():.3f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:41:26.500364Z","iopub.execute_input":"2026-01-16T07:41:26.501176Z","iopub.status.idle":"2026-01-16T07:41:26.918821Z","shell.execute_reply.started":"2026-01-16T07:41:26.501115Z","shell.execute_reply":"2026-01-16T07:41:26.917995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 3: Lead-wise submission builder (NO extra normalization)\n\nimport pandas as pd\nimport numpy as np\nimport os\nfrom tqdm import tqdm\n\nTEST_CSV_PATH = \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\nTEST_IMG_DIR = \"/kaggle/input/physionet-ecg-image-digitization/test\"\n\ntest_df = pd.read_csv(TEST_CSV_PATH)\n\ndef slice_to_rows(w, fs, lead):\n    # Lead II = 10s, others = 2.5s\n    dur = 10.0 if lead == \"II\" else 2.5\n    n = int(np.floor(fs * dur))\n    if len(w) == n:\n        return w\n    # resample exactly to expected rows\n    x_old = np.linspace(0, 1, len(w))\n    x_new = np.linspace(0, 1, n)\n    return np.interp(x_new, x_old, w)\n\nsubmission_rows = []\nbase_ids = test_df[\"id\"].unique()\n\nfor base_id in tqdm(base_ids):\n    img_path = os.path.join(TEST_IMG_DIR, f\"{base_id}.png\")\n    lead_waves = infer_lead_waveforms(img_path)\n\n    subset = test_df[test_df[\"id\"] == base_id]\n\n    for _, row in subset.iterrows():\n        lead = row[\"lead\"]\n        fs = row[\"fs\"]\n        n_rows = row[\"number_of_rows\"]\n\n        w = lead_waves[lead]\n        w = slice_to_rows(w, fs, lead)\n\n        assert len(w) == n_rows, f\"{base_id} {lead}: {len(w)} != {n_rows}\"\n\n        for i, val in enumerate(w):\n            submission_rows.append({\n                \"id\": f\"{base_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n\nsubmission_df = pd.DataFrame(submission_rows)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission_df.to_csv(out_path, index=False)\n\nprint(\"Saved:\", out_path)\nprint(\"Rows:\", len(submission_df))\nprint(submission_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T07:42:04.659871Z","iopub.execute_input":"2026-01-16T07:42:04.660206Z","iopub.status.idle":"2026-01-16T07:42:05.741928Z","shell.execute_reply.started":"2026-01-16T07:42:04.660177Z","shell.execute_reply":"2026-01-16T07:42:05.741214Z"}},"outputs":[],"execution_count":null}]}