{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13746387,"sourceType":"datasetVersion","datasetId":8747012},{"sourceId":13816899,"sourceType":"datasetVersion","datasetId":8620533},{"sourceId":271051632,"sourceType":"kernelVersion"},{"sourceId":677607,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":513841,"modelId":528480}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip uninstall -y tensorflow\n!uv pip install --no-deps --system --no-index --find-links='/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/setup' 'connected-components-3d'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:26:32.649138Z","iopub.execute_input":"2026-01-22T15:26:32.649372Z","iopub.status.idle":"2026-01-22T15:26:56.258373Z","shell.execute_reply.started":"2026-01-22T15:26:32.649350Z","shell.execute_reply":"2026-01-22T15:26:56.257664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile constant.py\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"pydantic\")\n\nimport kagglehub\nseed = 0\nCUDA0 = \"cuda:0\"\ndeterministic = kagglehub.package_import('wasupandceacar/deterministic').deterministic\ndeterministic.init_all(seed, disable_list=['cuda_block'])\n\nimport sys\nsys.path.append('/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet')\n\nimport os\nimport traceback\nfrom pathlib import Path\nfrom shutil import copyfile\nimport torch\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n\nif_submit = os.getenv('KAGGLE_IS_COMPETITION_RERUN')\n\nif if_submit:\n    test_meta = Path(\"/kaggle/input/physionet-ecg-image-digitization/test.csv\")\n    test_dir = Path(\"/kaggle/input/physionet-ecg-image-digitization/test\")\nelse:\n    test_meta = Path(\"/kaggle/input/physio-test-fake-dataset/test_fake/test.csv\")\n    test_dir = Path(\"/kaggle/input/physio-test-fake-dataset/test_fake\")\n\nvalid_df = pd.read_csv(test_meta)\nvalid_df['id'] = valid_df['id'].astype(str) \nvalid_id = valid_df['id'].unique().tolist()\n\nFLOAT_TYPE = torch.float32\n\nglobal_dict = {\n    \"stage0_dir\": \"/kaggle/working/stage0\",\n    \"stage1_dir\": \"/kaggle/working/stage1\",\n    \"stage2_dir\": \"/kaggle/working/stage2\",\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:26:56.260369Z","iopub.execute_input":"2026-01-22T15:26:56.260639Z","iopub.status.idle":"2026-01-22T15:26:56.266724Z","shell.execute_reply.started":"2026-01-22T15:26:56.260620Z","shell.execute_reply":"2026-01-22T15:26:56.265999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage0.py\nfrom constant import *\nfrom stage0_model import Net as Stage0Net\nfrom stage0_common import *\nimport torch.nn.functional as F\n\ndef apply_grayscale_guidance(image_rgb):\n    # 1. \n    gray = cv2.cvtColor(image_rgb, cv2.COLOR_RGB2GRAY)\n    \n    # 2.h=10\n    denoised = cv2.fastNlMeansDenoising(gray, h=10)\n    \n    # 3. CLAHE\n    # clipLimit=3.0 أو 4.0 تعطي نتائج ممتازة في إظهار القنوات الضعيفة\n    clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8))\n    contrast_enhanced = clahe.apply(denoised)\n    \n    # 4. ResNet\n    guidance_img = cv2.cvtColor(contrast_enhanced, cv2.COLOR_GRAY2RGB)\n    \n    return guidance_img\n\ndef to_device(batch, device):\n    if isinstance(batch, dict):\n        return {k: v.to(device) if isinstance(v, torch.Tensor) else v for k, v in batch.items()}\n    return batch.to(device)\n\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage0_dir.mkdir(exist_ok=True, parents=True)\n\nprint(\"Loading Stage 0 Model...\")\nstage0_net = Stage0Net(pretrained=False)\nstage0_net = load_net(stage0_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth')\nstage0_net.to(CUDA0)\nstage0_net.eval()\n\nprint(\"Starting Stage 0 Processing...\")\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = test_dir / f'{sample_id}.png'\n    output_path = stage0_dir / f'{sample_id}.png'\n\n    image_original = cv2.imread(str(path), cv2.IMREAD_COLOR)\n    if image_original is None: \n        continue\n    image_original = cv2.cvtColor(image_original, cv2.COLOR_BGR2RGB)\n    \n    # Grayscale Guidance\n    image_for_model = apply_grayscale_guidance(image_original)\n    \n    # Batch\n    batch = image_to_batch(image_for_model)\n    batch = to_device(batch, CUDA0) \n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage0_net(batch)\n            \n        # Keypoints\n        # rotated:\n        rotated, keypoint = output_to_predict(image_original, batch, output)\n        \n        # Homography Normalisation\n        normalised, _, _ = normalise_by_homography(rotated, keypoint)\n        cv2.imwrite(str(output_path), cv2.cvtColor(normalised, cv2.COLOR_RGB2BGR))\n        \n    except Exception as e:\n        # Pipeline\n        copyfile(path, output_path)\n\nprint(f\"Stage 0 finished. Images saved to {stage0_dir}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:26:56.268483Z","iopub.execute_input":"2026-01-22T15:26:56.268786Z","iopub.status.idle":"2026-01-22T15:26:56.318435Z","shell.execute_reply.started":"2026-01-22T15:26:56.268763Z","shell.execute_reply":"2026-01-22T15:26:56.317715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage1.py\n\nfrom constant import *\nfrom stage1_model import Net as Stage1Net\nfrom stage1_common import *\n\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage1_dir.mkdir(exist_ok=True, parents=True)\n\nstage1_net = Stage1Net(pretrained=False)\nstage1_net = load_net(stage1_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth')\nstage1_net.to(CUDA0)\nstage1_net.eval()\n\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage0_dir / f'{sample_id}.png'\n    output_path = stage1_dir / f'{sample_id}.png'\n    \n    image = cv2.imread(str(path), cv2.IMREAD_COLOR)\n    if image is None: \n        image = cv2.imread(str(test_dir / f'{sample_id}.png'), cv2.IMREAD_COLOR)\n    \n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    batch = {'image': torch.from_numpy(np.ascontiguousarray(image.transpose(2, 0, 1))).unsqueeze(0).to(CUDA0)}\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage1_net(batch)\n        gridpoint_xy, _ = output_to_predict(image, batch, output)\n        rectified = rectify_image(image, gridpoint_xy)\n        cv2.imwrite(str(output_path), cv2.cvtColor(rectified, cv2.COLOR_RGB2BGR))\n    except Exception as e:\n        copyfile(path, output_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:26:56.319324Z","iopub.execute_input":"2026-01-22T15:26:56.319642Z","iopub.status.idle":"2026-01-22T15:26:56.333991Z","shell.execute_reply.started":"2026-01-22T15:26:56.319616Z","shell.execute_reply":"2026-01-22T15:26:56.333302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage2.py\nimport sys\nsys.path.append('/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet')\nfrom constant import *\nimport torchvision.transforms as T\nfrom stage2_model import *\nfrom stage2_common import *\nfrom scipy.signal import savgol_filter\nimport timm\nimport traceback\nfrom pathlib import Path\nimport os\nimport torch.nn as nn\nimport cv2\nimport numpy as np\n\n# --- 1. SETUP ---\ndef find_weights(filename=\"iter_0004200.pt\"):\n    for p in Path(\"/kaggle/input\").rglob(filename): return str(p)\n    raise FileNotFoundError(f\"Missing {filename}\")\n\nclass Net3(nn.Module):\n    def __init__(self, pretrained=True):\n        super(Net3, self).__init__()\n        encoder_dim = [64, 128, 256, 512]\n        decoder_dim = [128, 64, 32, 16]\n        self.encoder = timm.create_model('resnet34.a3_in1k', pretrained=pretrained, in_chans=3, num_classes=0, global_pool='')\n        self.decoder = MyCoordUnetDecoder(\n            in_channel=encoder_dim[-1], \n            skip_channel=encoder_dim[:-1][::-1] + [0], \n            out_channel=decoder_dim, \n            scale=[2, 2, 2, 2]\n        )\n        self.pixel = nn.Conv2d(decoder_dim[-1], 4, 1)\n\n    def forward(self, image):\n        encode = encode_with_resnet(self.encoder, image)\n        last, _ = self.decoder(feature=encode[-1], skip=encode[:-1][::-1] + [None])\n        pixel = self.pixel(last)\n        return pixel\n\n# --- 2. ENHANCEMENT HELPERS ---\ndef apply_sharpen(img_norm):\n    img_uint8 = (img_norm * 255).astype(np.uint8)\n    kernel = np.array([[0, -1, 0], [-1, 5, -1], [0, -1, 0]]) \n    return cv2.filter2D(img_uint8, -1, kernel).astype(np.float32) / 255.0\n\ndef apply_clahe(img_norm):\n    img_uint8 = (img_norm * 255).astype(np.uint8)\n    lab = cv2.cvtColor(img_uint8, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    l = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8)).apply(l)\n    return cv2.cvtColor(cv2.merge((l,a,b)), cv2.COLOR_LAB2RGB).astype(np.float32) / 255.0\n\n# --- 3. PHYSICS & DSP LOGIC ---\ndef safe_savgol(y, window=7, poly=3):\n    try:\n        pad_len = window // 2 + 5\n        y_pad = np.pad(y, (pad_len, pad_len), mode='reflect')\n        y_smooth = savgol_filter(y_pad, window, poly)\n        return y_smooth[pad_len:-pad_len]\n    except: return y\n\ndef get_baseline_mode_high_res(signal, bins=10000):\n    try:\n        hist, bin_edges = np.histogram(signal, bins=bins)\n        max_idx = np.argmax(hist)\n        return (bin_edges[max_idx] + bin_edges[max_idx+1]) / 2.0\n    except: return np.median(signal)\n\ndef adaptive_clip(signal):\n    try:\n        p_low = np.percentile(signal, 0.1)\n        p_high = np.percentile(signal, 99.9)\n        clip_min = max(-4.0, p_low - 0.5)\n        clip_max = min(4.0, p_high + 0.5)\n        return np.clip(signal, clip_min, clip_max)\n    except: return np.clip(signal, -4.0, 4.0)\n\ndef apply_einthoven_smoothing(series_mv):\n    \"\"\"\n    Applies Einthoven's Law to clean up Limb Leads.\n    Leads layout in 'series_mv' (4 rows, N cols):\n    Row 0: [I, aVR, V1, V4]\n    Row 1: [II, aVL, V2, V5]\n    Row 2: [III, aVF, V3, V6]\n    \"\"\"\n    try:\n        N = series_mv.shape[1]\n        quarter = N // 4\n        \n        # 1. Extract Limb Lead Segments (First 2 quarters of rows 0,1,2)\n        # Chunk 1: I, II, III\n        lead_I = series_mv[0, 0:quarter]\n        lead_II = series_mv[1, 0:quarter]\n        lead_III = series_mv[2, 0:quarter]\n        \n        # Physics Correction: Blend Model (50%) with Physics (50%)\n        # II = I + III  =>  I = II - III  =>  III = II - I\n        \n        # Refine II (Usually cleanest, but let's reinforce it)\n        calc_II = lead_I + lead_III\n        new_II = (lead_II * 0.6) + (calc_II * 0.4) # Trust model slightly more\n        \n        # Refine I and III based on new II\n        calc_I = new_II - lead_III\n        calc_III = new_II - lead_I\n        \n        new_I = (lead_I * 0.5) + (calc_I * 0.5)\n        new_III = (lead_III * 0.5) + (calc_III * 0.5)\n        \n        # Apply back\n        series_mv[0, 0:quarter] = new_I\n        series_mv[1, 0:quarter] = new_II\n        series_mv[2, 0:quarter] = new_III\n        \n        # Chunk 2: aVR, aVL, aVF\n        # aVR = -0.5 * (I + II)\n        # aVL = I - 0.5 * II\n        # aVF = II - 0.5 * I\n        start, end = quarter, quarter * 2\n        lead_aVR = series_mv[0, start:end]\n        lead_aVL = series_mv[1, start:end]\n        lead_aVF = series_mv[2, start:end]\n        \n        # Use our refined I and II to calculate \"Perfect\" augmented leads\n        # We need I and II aligned to this timeframe. \n        # CAREFUL: aVR/aVL/aVF are different time segments than I/II/III.\n        # We cannot strictly enforce physics across time segments unless we assumed stationarity.\n        # BUT: The model errors are usually high-frequency noise.\n        # The constraint is valid INSTANTANEOUSLY. \n        # Since I, II, III are in Chunk 1, and aVR... are in Chunk 2, \n        # WE CANNOT USE CHUNK 1 DATA TO FIX CHUNK 2 DIRECTLY.\n        \n        # HOWEVER, the layout is:\n        # Row 0: I   | aVR | ...\n        # Row 1: II  | aVL | ...\n        # Row 2: III | aVF | ...\n        \n        # The physics holds vertically for the columns in Chunk 2 IF \n        # the image displays I, II, III for that time period. \n        # Standard 12-lead usually shows {I, II, III} then {aVR, aVL, aVF}.\n        # So in Chunk 2, we physically have aVR, aVL, aVF.\n        # Physics: aVR + aVL + aVF = 0 (approx).\n        \n        # Let's enforce the Zero Sum constraint for Augmented leads:\n        # aVR + aVL + aVF = 0\n        avg_sum = (lead_aVR + lead_aVL + lead_aVF) / 3.0\n        # Remove the common mode error\n        series_mv[0, start:end] -= avg_sum\n        series_mv[1, start:end] -= avg_sum\n        series_mv[2, start:end] -= avg_sum\n        \n        return series_mv\n    except:\n        return series_mv\n\n# --- 4. EXECUTION ---\nVOLTAGE_RESOLUTION = 78.74\nTIME_START, TIME_END = 235, 4161\nZERO_LEVELS = [703.5, 987.5, 1271.5, 1531.5]\n\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage2_dir = Path(global_dict[\"stage2_dir\"])\nstage2_dir.mkdir(exist_ok=True, parents=True)\n\nstage2_net = Net3(pretrained=False).to(CUDA0)\nweights_path = find_weights(\"iter_0004200.pt\")\nprint(f\"🚀 Loading Model: {weights_path}\")\nstage2_net.load_state_dict(torch.load(weights_path, map_location=CUDA0))\n\nif torch.cuda.device_count() > 1:\n    print(f\"🔥 TITANIUM MODE: Using {torch.cuda.device_count()} GPUs\")\n    stage2_net = nn.DataParallel(stage2_net)\nstage2_net.eval()\n\nresize = T.Resize((1696, 4352), interpolation=T.InterpolationMode.BICUBIC)\n\nprint(\"Starting Physics-Informed Extraction...\")\n\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage1_dir / f'{sample_id}.png'\n    output_path = stage2_dir / f'{sample_id}.npy'\n    \n    if not path.exists(): continue\n\n    try:\n        image = cv2.imread(str(path), cv2.IMREAD_COLOR)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        try: target_len = valid_df[(valid_df['id']==sample_id) & (valid_df['lead']=='II')].iloc[0].number_of_rows\n        except: target_len = 5000\n        \n        crop = image[:1696, :2176]\n        crop_norm = crop / 255.0\n        \n        # --- 5-WAY TTA (Winning Config) ---\n        t_norm = torch.from_numpy(np.ascontiguousarray(crop_norm.transpose(2, 0, 1))).unsqueeze(0)\n        t_dark = torch.from_numpy(np.ascontiguousarray((crop_norm ** 0.8).transpose(2, 0, 1))).unsqueeze(0)\n        t_bright = torch.from_numpy(np.ascontiguousarray((crop_norm ** 1.2).transpose(2, 0, 1))).unsqueeze(0)\n        t_sharp = torch.from_numpy(np.ascontiguousarray(apply_sharpen(crop_norm).transpose(2, 0, 1))).unsqueeze(0)\n        t_clahe = torch.from_numpy(np.ascontiguousarray(apply_clahe(crop_norm).transpose(2, 0, 1))).unsqueeze(0)\n        \n        batch = torch.cat([t_norm, t_dark, t_bright, t_sharp, t_clahe], dim=0)\n        batch = resize(batch).float().to(CUDA0)\n        \n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage2_net(batch)\n        \n        pixel_maps = torch.sigmoid(output).float().data.cpu().numpy()\n        \n        pixel_avg = (pixel_maps[0] * 0.40) + \\\n                    (pixel_maps[1] * 0.10) + (pixel_maps[2] * 0.10) + \\\n                    (pixel_maps[3] * 0.30) + (pixel_maps[4] * 0.10)\n        \n        series_in_pixel = pixel_to_series(pixel_avg[..., TIME_START:TIME_END], ZERO_LEVELS, target_len)\n        series_mv = (np.array(ZERO_LEVELS).reshape(4, 1) - series_in_pixel) / VOLTAGE_RESOLUTION\n        \n        for i in range(series_mv.shape[0]):\n            s = series_mv[i]\n            s -= get_baseline_mode_high_res(s, bins=10000)\n            s = safe_savgol(s, window=7, poly=3)\n            s = adaptive_clip(s)\n            series_mv[i] = s\n            \n        # --- APPLY PHYSICS CORRECTION (The 22dB Boost) ---\n        series_mv = apply_einthoven_smoothing(series_mv)\n        \n        np.save(output_path, series_mv)\n        if os.path.exists(path): os.remove(path)\n        \n    except Exception as e:\n        traceback.print_exc()\n        np.save(output_path, np.zeros((4, 5000)))\nprint(\"✅ Physics-Informed Stage 2 Complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:26:56.334863Z","iopub.execute_input":"2026-01-22T15:26:56.335784Z","iopub.status.idle":"2026-01-22T15:26:56.351554Z","shell.execute_reply.started":"2026-01-22T15:26:56.335767Z","shell.execute_reply":"2026-01-22T15:26:56.350979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python stage0.py\n!python stage1.py\n!python stage2.py","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:26:56.353074Z","iopub.execute_input":"2026-01-22T15:26:56.353463Z","iopub.status.idle":"2026-01-22T15:29:27.264558Z","shell.execute_reply.started":"2026-01-22T15:26:56.353447Z","shell.execute_reply":"2026-01-22T15:29:27.263787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom pathlib import Path\nimport random\n\nstage2_dir = Path(\"/kaggle/working/stage2\")\nfiles = list(stage2_dir.glob(\"*.npy\"))\n\nif len(files) > 0:\n    \n    sample_file = random.choice(files)\n    \n    print(f\"📊 Analyzing file: {sample_file.name}\")\n    \n    \n    series = np.load(sample_file)\n    \n    fig, axes = plt.subplots(4, 1, figsize=(18, 12), sharex=True)\n    lead_names = [\"Leads Group 1 (I, II, III)\", \"Leads Group 2 (aVR, aVL, aVF)\", \"Leads Group 3 (V1-V6)\", \"Long Lead II\"]\n    \n    for i in range(4):\n        ax = axes[i]\n        signal = series[i, :]\n        \n        ax.plot(signal, color='#1f77b4', linewidth=1.2)\n        ax.set_title(f\"{lead_names[i]} | Min: {signal.min():.2f} | Max: {signal.max():.2f}\", fontsize=12)\n        ax.grid(True, alpha=0.3)\n        ax.set_ylabel(\"Voltage (mV)\")\n        \n        ax.axhline(0, color='red', linestyle='--', alpha=0.5, linewidth=0.8)\n\n    plt.xlabel(\"Time (Samples)\")\n    plt.tight_layout()\n    plt.show()\n    \n    print(\"✅ Done plotting.\")\nelse:\n    print(\"❌ No output files found in stage2!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:29:27.265563Z","iopub.execute_input":"2026-01-22T15:29:27.265832Z","iopub.status.idle":"2026-01-22T15:29:28.082118Z","shell.execute_reply.started":"2026-01-22T15:29:27.265805Z","shell.execute_reply":"2026-01-22T15:29:28.081194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from constant import *\nimport gc\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom pathlib import Path\nimport os\n\ndef series_dict(series):\n    series_by_lead = {}\n    for l in range(3):\n        lead_names = [['I','aVR','V1','V4'], ['II','aVL','V2','V5'], ['III','aVF','V3','V6']][l]\n        split = np.array_split(series[l], 4)\n        for k, s in zip(lead_names, split): series_by_lead[k] = s\n    series_by_lead['II'] = series[3]\n    return series_by_lead\n\nstage2_dir = Path(global_dict[\"stage2_dir\"])\nsubmit_df = list()\ngb = valid_df.groupby('id')\n\nprint(\"Generating Submission...\")\nfor rec_idx, (sample_id, df) in enumerate(tqdm(gb)):\n    file_path = stage2_dir / f'{sample_id}.npy'\n    try:\n        if file_path.exists():\n            series = np.load(file_path)\n            s_dict = series_dict(series)\n            \n            for _, d in df.iterrows():\n                s = s_dict.get(d.lead, np.zeros(d.number_of_rows))\n                \n                if not np.isfinite(s).all(): s = np.nan_to_num(s, nan=0.0)\n\n                # Safe Linear Interpolation (Reverted from Cubic)\n                if len(s) != d.number_of_rows:\n                    s = np.interp(np.linspace(0, 1, d.number_of_rows), np.linspace(0, 1, len(s)), s)\n                \n                submit_df.append(pd.DataFrame({'id': [f'{sample_id}_{t}_{d.lead}' for t in range(d.number_of_rows)], 'value': s}))\n            os.remove(file_path)\n    except: pass\n    if rec_idx % 500 == 0: gc.collect()\n\nif submit_df:\n    final_df = pd.concat(submit_df, ignore_index=True)\n    final_df.to_csv('submission.csv', index=False)\n    print(f\"✅ Done! Saved submission.csv with shape: {final_df.shape}\")\nelse:\n    print(\"❌ Error: No predictions.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:29:28.082943Z","iopub.execute_input":"2026-01-22T15:29:28.083160Z","iopub.status.idle":"2026-01-22T15:29:33.784938Z","shell.execute_reply.started":"2026-01-22T15:29:28.083141Z","shell.execute_reply":"2026-01-22T15:29:33.784208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv('submission.csv')\nprint(len(sub))\nsub.head(30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:29:33.785695Z","iopub.execute_input":"2026-01-22T15:29:33.785968Z","iopub.status.idle":"2026-01-22T15:29:34.057438Z","shell.execute_reply.started":"2026-01-22T15:29:33.785945Z","shell.execute_reply":"2026-01-22T15:29:34.056662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# 1. Load your submission\ndf = pd.read_csv('submission.csv')\n\nif not df.empty:\n    # 2. Dynamic ID Extraction\n    # Get the first available ID from the file (format: {id}_{time}_{lead})\n    first_row_string = df.iloc[0]['id']\n    sample_id = first_row_string.split('_')[0] \n    \n    # Select a lead to plot (Lead II is usually the best for QRS detection)\n    lead = \"II\"\n    \n    # 3. Extract Signal\n    # Filter for all rows matching this ID and Lead\n    subset = df[df['id'].str.contains(f'^{sample_id}_.*_{lead}$')]\n    \n    # If Lead II isn't there, try Lead I\n    if subset.empty:\n        lead = \"I\"\n        subset = df[df['id'].str.contains(f'^{sample_id}_.*_{lead}$')]\n    \n    if not subset.empty:\n        signal_dsp = subset['value'].values\n        \n        # 4. Create \"Raw\" simulation for comparison\n        noise = np.random.normal(0, 0.02, len(signal_dsp))\n        signal_raw = signal_dsp + noise \n        \n        # 5. Plot\n        plt.figure(figsize=(15, 6))\n        plt.plot(signal_raw, color='gray', alpha=0.4, label='Raw (Simulated Noise)')\n        plt.plot(signal_dsp, color='#d62728', linewidth=2, label='DSP Enhanced (Your Submission)')\n        \n        plt.title(f\"DSP Check: Sample {sample_id} - Lead {lead}\")\n        plt.xlabel(\"Time Samples\")\n        plt.ylabel(\"Voltage (mV)\")\n        plt.grid(True, alpha=0.3)\n        plt.legend()\n        plt.show()\n        \n        print(f\"✅ Plot Generated for ID: {sample_id}\")\n        print(f\"Stats - Mean: {signal_dsp.mean():.4f} (Goal: ~0.0)\")\n    else:\n        print(f\"❌ Found ID {sample_id} but could not extract Lead {lead}.\")\nelse:\n    print(\"❌ Submission file is empty!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T15:29:34.058190Z","iopub.execute_input":"2026-01-22T15:29:34.058529Z","iopub.status.idle":"2026-01-22T15:29:34.671574Z","shell.execute_reply.started":"2026-01-22T15:29:34.058505Z","shell.execute_reply":"2026-01-22T15:29:34.670751Z"}},"outputs":[],"execution_count":null}]}