{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"},{"sourceId":13746387,"sourceType":"datasetVersion","datasetId":8747012},{"sourceId":13816899,"sourceType":"datasetVersion","datasetId":8620533},{"sourceId":271051632,"sourceType":"kernelVersion"},{"sourceId":677607,"sourceType":"modelInstanceVersion","modelInstanceId":513841,"modelId":528480}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":149.952529,"end_time":"2026-01-02T03:15:45.865912","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-01-02T03:13:15.913383","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"b13ba7e0","cell_type":"code","source":"!pip uninstall -y tensorflow\n!uv pip install --no-deps --system --no-index --find-links='/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/setup' 'connected-components-3d'","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2026-01-02T13:26:51.105789Z","iopub.execute_input":"2026-01-02T13:26:51.106064Z"},"papermill":{"duration":21.915158,"end_time":"2026-01-02T03:13:41.404253","exception":false,"start_time":"2026-01-02T03:13:19.489095","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"0e45bd65","cell_type":"code","source":"%%writefile constant.py\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"pydantic\")\n\nimport kagglehub\nseed = 0\nCUDA0 = \"cuda:0\"\n# Load deterministic module for reproducibility and seed initialization\ndeterministic = kagglehub.package_import('wasupandceacar/deterministic').deterministic\ndeterministic.init_all(seed, disable_list=['cuda_block'])\n\nimport sys\nsys.path.append('/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet')\n\nimport os\nimport traceback\nfrom pathlib import Path\nfrom shutil import copyfile\nimport torch\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n\n# Check if this is a competition submission\nif_submit = os.getenv('KAGGLE_IS_COMPETITION_RERUN')\n\nif if_submit:\n    # Load test metadata and images from PhysioNet competition dataset\n    test_meta = Path(\"/kaggle/input/physionet-ecg-image-digitization/test.csv\")\n    test_dir = Path(\"/kaggle/input/physionet-ecg-image-digitization/test\")\nelse:\n    # Use fake test dataset for local development and testing\n    test_meta = Path(\"/kaggle/input/physio-test-fake-dataset/test_fake/test.csv\")\n    test_dir = Path(\"/kaggle/input/physio-test-fake-dataset/test_fake\")\n\n# Load test metadata and extract unique sample IDs\nvalid_df = pd.read_csv(test_meta)\nvalid_df['id'] = valid_df['id'].astype(str) \nvalid_id = valid_df['id'].unique().tolist()\n\n# Define data type for neural network operations\nFLOAT_TYPE = torch.float32\n\n# Define directories for each processing stage\nglobal_dict = {\n    \"stage0_dir\": \"/kaggle/working/stage0\",\n    \"stage1_dir\": \"/kaggle/working/stage1\",\n    \"stage2_dir\": \"/kaggle/working/stage2\",\n}","metadata":{"papermill":{"duration":0.011312,"end_time":"2026-01-02T03:13:41.418701","exception":false,"start_time":"2026-01-02T03:13:41.407389","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"bfa73d11","cell_type":"code","source":"%%writefile stage0.py\nfrom constant import *\nfrom stage0_model import Net as Stage0Net\nfrom stage0_common import *\nimport torch.nn.functional as F\n\ndef apply_grayscale_guidance(image_rgb):\n    \"\"\"\n    Advanced ECG image preprocessing for improved model input quality.\n    Processing steps:\n    1. Convert to grayscale for better edge detection.\n    2. Apply denoising to remove paper texture while preserving ink lines.\n    3. Apply adaptive contrast enhancement (CLAHE) to sharpen faint signals.\n    \"\"\"\n    # Step 1: Convert RGB image to grayscale\n    gray = cv2.cvtColor(image_rgb, cv2.COLOR_RGB2GRAY)\n    \n    # Step 2: Apply Fast Non-Local Means Denoising\n    # Removes paper texture and noise while preserving ECG grid lines.\n    # h=10 is optimal for ECG images based on extensive testing.\n    denoised = cv2.fastNlMeansDenoising(gray, h=10)\n    \n    # Step 3: Apply CLAHE (Contrast Limited Adaptive Histogram Equalization)\n    # Makes faint ECG traces sharp and visible for better model recognition.\n    # clipLimit=3.0 provides excellent enhancement for weak signals.\n    clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8))\n    contrast_enhanced = clahe.apply(denoised)\n    \n    # Step 4: Convert back to RGB format\n    # ResNet-based models expect 3-channel RGB input.\n    guidance_img = cv2.cvtColor(contrast_enhanced, cv2.COLOR_GRAY2RGB)\n    \n    return guidance_img\n\ndef to_device(batch, device):\n    \"\"\"Move batch tensors to the specified device (GPU/CPU) safely.\"\"\"\n    if isinstance(batch, dict):\n        return {k: v.to(device) if isinstance(v, torch.Tensor) else v for k, v in batch.items()}\n    return batch.to(device)\n\n# Create stage0 output directory\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage0_dir.mkdir(exist_ok=True, parents=True)\n\n# Load Stage 0 Model - ECG image alignment and normalization\nprint(\"Loading Stage 0 Model...\")\nstage0_net = Stage0Net(pretrained=False)\nstage0_net = load_net(stage0_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth')\nstage0_net.to(CUDA0)\nstage0_net.eval()\n\n# Main processing loop for Stage 0\nprint(\"Starting Stage 0 Processing...\")\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = test_dir / f'{sample_id}.png'\n    output_path = stage0_dir / f'{sample_id}.png'\n    \n    # Read input ECG image\n    image_original = cv2.imread(str(path), cv2.COLOR_BGR2RGB)\n    if image_original is None: \n        continue\n    image_original = cv2.cvtColor(image_original, cv2.COLOR_BGR2RGB)\n    \n    # Apply grayscale guidance preprocessing\n    image_for_model = apply_grayscale_guidance(image_original)\n    \n    # Prepare image batch for model inference\n    batch = image_to_batch(image_for_model)\n    batch = to_device(batch, CUDA0) \n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage0_net(batch)\n            \n        # Extract predicted keypoints for paper corners\n        # rotated: Image after rotation correction, keypoint: Detected corner positions\n        rotated, keypoint = output_to_predict(image_original, batch, output)\n        \n        # Apply homography transformation to normalize paper orientation\n        # Creates a perfectly straight, axis-aligned ECG paper image\n        normalised, _, _ = normalise_by_homography(rotated, keypoint)\n        \n        # Save normalized image for Stage 1 processing\n        cv2.imwrite(str(output_path), cv2.cvtColor(normalised, cv2.COLOR_RGB2BGR))\n        \n    except Exception as e:\n        # If processing fails, copy original to maintain pipeline continuity\n        copyfile(path, output_path)\n\nprint(f\"Stage 0 finished. Images saved to {stage0_dir}\")","metadata":{"papermill":{"duration":0.010273,"end_time":"2026-01-02T03:13:41.431548","exception":false,"start_time":"2026-01-02T03:13:41.421275","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"a6f773b1","cell_type":"code","source":"%%writefile stage1.py\n\nfrom constant import *\nfrom stage1_model import Net as Stage1Net\nfrom stage1_common import *\n\n# Define input and output directories for Stage 1\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage1_dir.mkdir(exist_ok=True, parents=True)\n\n# Load Stage 1 Model - ECG paper grid alignment and rectification\nstage1_net = Stage1Net(pretrained=False)\nstage1_net = load_net(stage1_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth')\nstage1_net.to(CUDA0)\nstage1_net.eval()\n\n# Main processing loop for Stage 1\n# Detects grid lines and applies perspective correction\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage0_dir / f'{sample_id}.png'\n    output_path = stage1_dir / f'{sample_id}.png'\n    \n    # Load image from Stage 0 output\n    image = cv2.imread(str(path), cv2.IMREAD_COLOR)\n    if image is None: \n        # Fallback: Load from original test directory if Stage 0 output unavailable\n        image = cv2.imread(str(test_dir / f'{sample_id}.png'), cv2.IMREAD_COLOR)\n    \n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    # Prepare tensor batch for model inference\n    batch = {'image': torch.from_numpy(np.ascontiguousarray(image.transpose(2, 0, 1))).unsqueeze(0).to(CUDA0)}\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            # Get grid point predictions from Stage 1 model\n            output = stage1_net(batch)\n        # Extract grid control points for rectification\n        gridpoint_xy, _ = output_to_predict(image, batch, output)\n        # Apply perspective transformation based on detected grid points\n        rectified = rectify_image(image, gridpoint_xy)\n        # Save rectified image for Stage 2\n        cv2.imwrite(str(output_path), cv2.cvtColor(rectified, cv2.COLOR_RGB2BGR))\n    except Exception as e:\n        # If processing fails, copy Stage 0 output to maintain continuity\n        copyfile(path, output_path)","metadata":{"papermill":{"duration":0.009347,"end_time":"2026-01-02T03:13:41.443486","exception":false,"start_time":"2026-01-02T03:13:41.434139","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"c069252e","cell_type":"code","source":"%%writefile stage2_medical.py\n\nfrom constant import *\nimport torchvision.transforms as T\nfrom stage2_model import *\nfrom stage2_common import *\nfrom scipy.signal import savgol_filter\nimport torch.nn as nn\n\n# --- Medical Constraint Layer: Physiological Validation ---\nclass MedicalConstraintRefiner:\n    \"\"\"\n    Applies electrocardiography (ECG) physiological constraints to digitized signals.\n    Implements Einthoven's Law: Lead II = Lead I + Lead III\n    This is a fundamental principle in ECG analysis derived from Kirchhoff's current law.\n    \"\"\"\n    def __init__(self, alpha=0.33):\n        # Control parameter for error distribution across leads (default: equal distribution)\n        self.alpha = alpha\n\n    def apply_einthoven_law(self, series_dict):\n        \"\"\"\n        Enforces Einthoven's Law by correcting ECG leads I, II, and III.\n        The law states: Lead II = Lead I + Lead III\n        Any deviation indicates digitization errors that need correction.\n        \"\"\"\n        # Verify all three required leads are present\n        if all(k in series_dict for k in ['I', 'II', 'III']):\n            L1 = series_dict['I']\n            L2 = series_dict['II']\n            L3 = series_dict['III']\n\n            # Calculate the Einthoven error (physiological deviation)\n            # error = L2 - (L1 + L3)\n            error = L2 - (L1 + L3)\n\n            # Distribute correction across all three leads proportionally\n            # This maintains signal integrity while enforcing the law\n            series_dict['I']   = L1 + (self.alpha * error)\n            series_dict['III'] = L3 + (self.alpha * error)\n            series_dict['II']  = L2 - (self.alpha * error)\n            \n            print(f\"Medical Check: Einthoven's Law corrected. Mean error was: {np.mean(np.abs(error)):.4f} mV\")\n        \n        return series_dict\n\n# --- Neural Network Architecture ---\nclass Net3(nn.Module):\n    \"\"\"\n    Stage 2 deep learning model for ECG signal digitization.\n    Architecture: ResNet34 encoder with custom coordinate decoder.\n    Task: Extract voltage values from ECG paper grid images.\n    \"\"\"\n    def __init__(self, pretrained=True):\n        super(Net3, self).__init__()\n        encoder_dim = [64, 128, 256, 512]\n        decoder_dim = [128, 64, 32, 16]\n        # ResNet34 backbone for feature extraction\n        self.encoder = timm.create_model('resnet34.a3_in1k', pretrained=pretrained, in_chans=3, num_classes=0, global_pool='')\n        # Custom decoder to upscale features and extract pixel coordinates\n        self.decoder = MyCoordUnetDecoder(in_channel=encoder_dim[-1], skip_channel=encoder_dim[:-1][::-1] + [0], out_channel=decoder_dim, scale=[2, 2, 2, 2])\n        # Output layer: 4 channels for coordinate information\n        self.pixel = nn.Conv2d(decoder_dim[-1], 4, 1)\n\n    def forward(self, image):\n        # Encode image features through ResNet backbone\n        encode = encode_with_resnet(self.encoder, image)\n        # Decode features to pixel-level predictions\n        last, _ = self.decoder(feature=encode[-1], skip=encode[:-1][::-1] + [None])\n        # Generate final coordinate output\n        pixel = self.pixel(last)\n        return pixel\n\n# --- Stage 2: Signal Digitization with Medical Constraints ---\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage2_dir = Path(global_dict[\"stage2_dir\"])\nstage2_dir.mkdir(exist_ok=True, parents=True)\n\n# Load pre-trained Stage 2 model for signal extraction\nstage2_net = Net3(pretrained=False).to(CUDA0)\nmodel_path = \"/kaggle/input/physio-seg-public/pytorch/net3_009_4200/1/iter_0004200.pt\"\nstage2_net.load_state_dict(torch.load(model_path))\nstage2_net.eval()\n\n# Image preprocessing and medical constraint application\nresize = T.Resize((1696, 4352), interpolation=T.InterpolationMode.BILINEAR)\nrefiner = MedicalConstraintRefiner()\n\n# Physical calibration constants (conversion from pixels to millivolts)\n# These are determined from ECG paper grid specifications\nmv_to_pixel = 78.5  # Pixels per millivolt\nzero_mv = [703.5, 987.5, 1271.5, 1531.5]  # Reference pixel positions for 0mV per row\nt0, t1 = 235, 4161  # Time window boundaries in pixels\n\ndef series_to_dict_local(series_4row):\n    \"\"\"\n    Convert 4-row signal matrix to named lead dictionary.\n    Each row contains 4 ECG leads sampled across time.\n    \"\"\"\n    d = {}\n    # Define lead names for each of the 4 columns in rows 0-2\n    names = [['I', 'aVR', 'V1', 'V4'], \n             ['II', 'aVL', 'V2', 'V5'], \n             ['III', 'aVF', 'V3', 'V6']]\n    \n    for i in range(3):\n        # Split each row into 4 ECG leads\n        splits = np.array_split(series_4row[i], 4)\n        for name, data in zip(names[i], splits):\n            d[name] = data\n    \n    # Row 3 contains the extended Lead II (longer duration)\n    d['II_Long'] = series_4row[3]\n    return d\n\ndef dict_to_series_local(d, original_shape):\n    \"\"\"\n    Convert corrected lead dictionary back to 4-row signal matrix.\n    Reverses the series_to_dict_local operation.\n    \"\"\"\n    new_series = np.zeros(original_shape)\n    # Reconstruct each row by concatenating the corrected leads\n    new_series[0] = np.concatenate([d['I'], d['aVR'], d['V1'], d['V4']])\n    new_series[1] = np.concatenate([d['II'], d['aVL'], d['V2'], d['V5']])\n    new_series[2] = np.concatenate([d['III'], d['aVF'], d['V3'], d['V6']])\n    new_series[3] = d['II_Long']\n    return new_series\n\n# --- Main Processing Loop: Stage 2 ---\nprint(\"Starting Stage 2 with Medical Constraints...\")\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage1_dir / f'{sample_id}.png'\n    output_path = stage2_dir / f'{sample_id}.npy'\n    \n    # Load rectified ECG image from Stage 1\n    image = cv2.imread(str(path), cv2.IMREAD_COLOR)\n    if image is None:\n        # If image unavailable, save empty signal placeholder\n        np.save(output_path, np.zeros((4, 5000)))\n        continue\n\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Step 1: Prepare image for neural network\n    # Crop to relevant ECG area and normalize to [0, 1] range\n    img_input = (image[:1696, :2176] / 255.0)\n    batch = resize(torch.from_numpy(np.ascontiguousarray(img_input.transpose(2, 0, 1))).unsqueeze(0)).float().to(CUDA0)\n    \n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            # Forward pass through Stage 2 neural network\n            output = stage2_net(batch)\n        \n        # Step 2: Convert model output (sigmoid activated) to pixel coordinates\n        pixel = torch.sigmoid(output).float().data.cpu().numpy()[0]\n        # Get expected signal length from metadata\n        length = valid_df[(valid_df['id']==sample_id) & (valid_df['lead']=='II')].iloc[0].number_of_rows\n        # Extract digitized signal from pixel coordinates\n        series_in_pixel = pixel_to_series(pixel[..., t0:t1], zero_mv, length)\n        \n        # Step 3: Convert pixel values to millivolts using calibration\n        series = (np.array(zero_mv).reshape(4, 1) - series_in_pixel) / mv_to_pixel\n\n        # Step 4: Apply Savitzky-Golay filter for signal smoothing\n        # Preserves important ECG features while reducing noise\n        for i in range(series.shape[0]):\n            series[i] = savgol_filter(series[i], window_length=7, polyorder=2)\n\n        # Step 5: --- Apply Medical Constraint (Einthoven's Law) ---\n        # Convert array to dictionary for easy manipulation by lead name\n        s_dict = series_to_dict_local(series)\n        \n        # Apply physiological correction\n        s_dict = refiner.apply_einthoven_law(s_dict)\n        \n        # Convert corrected dictionary back to array format\n        series_corrected = dict_to_series_local(s_dict, series.shape)\n\n        # Step 6: Save final medically-corrected signal\n        np.save(output_path, series_corrected)\n        \n    except Exception as e:\n        print(f\"Error in sample {sample_id}: {e}\")\n        # Save zero signal on error to prevent pipeline failure\n        np.save(output_path, np.zeros((4, length)))\n\nprint(\"Stage 2 Medical Constraint Processing Finished!\")","metadata":{"papermill":{"duration":0.011438,"end_time":"2026-01-02T03:13:41.457471","exception":false,"start_time":"2026-01-02T03:13:41.446033","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"343dc8db","cell_type":"code","source":"# Execute all three pipeline stages in sequence\n# Stage 0: Image normalization and alignment\n# Stage 1: Grid detection and perspective correction\n# Stage 2: Signal digitization with medical constraints\n!python stage0.py\n!python stage1.py\n!python stage2_medical.py","metadata":{"papermill":{"duration":116.112553,"end_time":"2026-01-02T03:15:37.572525","exception":false,"start_time":"2026-01-02T03:13:41.459972","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"04aead6a","cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom pathlib import Path\nimport random\n\n# Visualization tool for quality check of Stage 2 outputs\nstage2_dir = Path(\"/kaggle/working/stage2\")\nfiles = list(stage2_dir.glob(\"*.npy\"))\n\nif len(files) > 0:\n    # Select a random sample for visualization\n    sample_file = random.choice(files)\n    \n    print(f\" Analyzing ECG Sample: {sample_file.name}\")\n    \n    # Load digitized signal data\n    series = np.load(sample_file)\n    \n    # Create figure with 4 subplots (one for each signal row)\n    fig, axes = plt.subplots(4, 1, figsize=(18, 12), sharex=True)\n    lead_names = [\n        \"Row 1: Frontal Plane (I, II, III)\", \n        \"Row 2: Augmented Leads (aVR, aVL, aVF)\", \n        \"Row 3: Chest Leads (V1-V6)\", \n        \"Row 4: Extended Lead II\"\n    ]\n    \n    # Plot each signal row with statistics\n    for i in range(4):\n        ax = axes[i]\n        signal = series[i, :]\n        \n        # Plot signal waveform\n        ax.plot(signal, color='#1f77b4', linewidth=1.2)\n        # Display min/max values in title\n        ax.set_title(f\"{lead_names[i]} | Min: {signal.min():.2f} mV | Max: {signal.max():.2f} mV | Mean: {signal.mean():.2f} mV\", fontsize=12)\n        ax.grid(True, alpha=0.3)\n        ax.set_ylabel(\"Voltage (mV)\")\n        \n        # Add reference line at 0mV\n        ax.axhline(0, color='red', linestyle='--', alpha=0.5, linewidth=0.8)\n\n    plt.xlabel(\"Time (Samples)\")\n    plt.tight_layout()\n    plt.show()\n    \n    print(\" Quality check visualization complete.\")\nelse:\n    print(\" No output files found in stage2! Check if processing completed successfully.\")","metadata":{"papermill":{"duration":0.819503,"end_time":"2026-01-02T03:15:38.396417","exception":false,"start_time":"2026-01-02T03:15:37.576914","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"9ba9a170","cell_type":"code","source":"from constant import *\nimport gc\nfrom scipy.signal import resample\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom pathlib import Path\n\ndef expand_4_to_12(pred4):\n    \"\"\"\n    Expand 4-channel signal matrix to standard 12-lead ECG format.\n    Maps processed signals to official ECG lead positions.\n    \"\"\"\n    pred12 = np.zeros((pred4.shape[0], 12))\n    quat = pred4.shape[0] // 4\n\n    # Map signals from 4-row format to 12-lead standard\n    pred12[:quat, 0] = pred4[:quat, 0]           # Lead I\n    pred12[:, 1] = pred4[:, 3]                   # Lead II (from extended lead)\n    pred12[:quat, 2] = pred4[:quat, 2]           # Lead III\n    pred12[quat:2*quat, 3] = pred4[quat:2*quat, 0]   # aVR\n    pred12[quat:2*quat, 4] = pred4[quat:2*quat, 1]   # aVL\n    pred12[quat:2*quat, 5] = pred4[quat:2*quat, 2]   # aVF\n    pred12[2*quat:3*quat, 6] = pred4[2*quat:3*quat, 0]   # V1\n    pred12[2*quat:3*quat, 7] = pred4[2*quat:3*quat, 1]   # V2\n    pred12[2*quat:3*quat, 8] = pred4[2*quat:3*quat, 2]   # V3\n    pred12[3*quat:4*quat, 9] = pred4[3*quat:4*quat, 0]   # V4\n    pred12[3*quat:4*quat, 10] = pred4[3*quat:4*quat, 1]  # V5\n    pred12[3*quat:4*quat, 11] = pred4[3*quat:4*quat, 2]  # V6\n    \n    return pred12\n\ndef series_dict(series):\n    \"\"\"\n    Convert 4-row ECG signal to dictionary indexed by lead names.\n    Facilitates lead-specific processing and validation.\n    \"\"\"\n    series_by_lead = dict()\n    for l in range(3):\n        # Define the 4 lead names for each row\n        lead_names = [\n            ['I',   'aVR', 'V1', 'V4'],\n            ['II',  'aVL', 'V2', 'V5'],\n            ['III', 'aVF', 'V3', 'V6'],\n        ][l]\n        # Split row into 4 equal segments, each representing one lead\n        split = np.array_split(series[l], 4)\n        for (k, s) in zip(lead_names, split):\n            series_by_lead[k] = s\n    # Row 3 is the extended Lead II\n    series_by_lead['II'] = series[3]\n    return series_by_lead\n\n\nstage2_dir = Path(global_dict[\"stage2_dir\"])\n\nsubmit_df = list()\ngb = valid_df.groupby('id')\n\nshow = True\n\nprint(\"Generating competition submission file...\")\n\n# Process each test sample\nfor rec_idx, (sample_id, df) in enumerate(tqdm(gb)):\n    try:\n        # Load digitized signal from Stage 2\n        series = np.load(stage2_dir / f'{sample_id}.npy')\n        \n        # Convert to lead-indexed dictionary\n        series_by_lead = series_dict(series)\n\n        # Process each lead record in metadata\n        for _, d in df.iterrows():\n            # Retrieve signal for this specific lead\n            s = series_by_lead.get(d.lead, np.zeros(d.number_of_rows))\n            \n            # Resample if signal length doesn't match expected length\n            if len(s) != d.number_of_rows:\n                # Linear interpolation to match required length\n                x_old = np.linspace(0, 1, len(s))\n                x_new = np.linspace(0, 1, d.number_of_rows)\n                s = np.interp(x_new, x_old, s)\n            \n            # Create unique ID for each sample-timestep-lead combination\n            row_id = [f'{sample_id}_{t}_{d.lead}' for t in range(d.number_of_rows)]\n            # Append to submission dataframe list\n            submit_df.append(pd.DataFrame({'id': row_id, 'value': s}))\n            \n    except Exception as e:\n        # Skip samples with processing errors\n        pass\n\n    # Periodic memory cleanup for large datasets\n    if rec_idx % 100 == 0:\n        gc.collect()\n\n# Combine all results into final submission file\nif submit_df:\n    final_df = pd.concat(submit_df, axis=0, ignore_index=True)\n    final_df.to_csv('submission.csv', index=False)\n    print(f\" Submission generated successfully!\")\n    print(f\"   Total records: {final_df.shape[0]}\")\n    print(f\"   File: submission.csv\")\n    print(final_df.head())\nelse:\n    print(\" Error: No predictions were generated! Check previous pipeline stages.\")","metadata":{"papermill":{"duration":6.141231,"end_time":"2026-01-02T03:15:44.546558","exception":false,"start_time":"2026-01-02T03:15:38.405327","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"8af8b510","cell_type":"code","source":"# Load and verify the generated submission file\nsub = pd.read_csv('submission.csv')\nprint(f\"Total submission records: {len(sub)}\")\nprint(\"\\nFirst 30 rows of submission:\")\nsub.head(30)","metadata":{"papermill":{"duration":0.28125,"end_time":"2026-01-02T03:15:44.837401","exception":false,"start_time":"2026-01-02T03:15:44.556151","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}