{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31239,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport glob\n\n# --- CONFIGURATION ---\n# We will try to find the specific \"hard\" image from your plot (Sample 1)\n# Based on your image label, the ID is likely '1015663939'\nTARGET_ID_PARTIAL = '1015663939' \nSEARCH_DIR = '/kaggle/input/physionet-ecg-image-digitization/train'  # Adjust if your path is different\n\ndef solve_shadows_with_adaptive_threshold():\n    # 1. Find the specific \"hard\" image file\n    # We search recursively for the filename part\n    files = glob.glob(os.path.join(SEARCH_DIR, '**', f'*{TARGET_ID_PARTIAL}*.png'), recursive=True)\n    \n    if not files:\n        print(f\"Could not find image with ID {TARGET_ID_PARTIAL}\")\n        # Fallback to any random image if specific one not found\n        files = glob.glob(os.path.join(SEARCH_DIR, '**', '*.png'), recursive=True)\n        \n    img_path = files[1]\n    print(f\"Processing: {img_path}\")\n\n    # 2. Load Image\n    original = cv2.imread(img_path)\n    # Convert to Grayscale (simplifies data to just 'intensity')\n    gray = cv2.cvtColor(original, cv2.COLOR_BGR2GRAY)\n\n    # 3. Approach A: Global Thresholding (The OLD way - Fails on shadows)\n    # We use Otsu's method again to show why it fails\n    _, global_thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n\n    # 4. Approach B: Adaptive Thresholding (The NEW way - Fixes shadows)\n    # param 1: source image\n    # param 2: max value (255 = white)\n    # param 3: method (Gaussian looks at neighbors with a weighted average)\n    # param 4: type (Binary Inverse makes the signal WHITE, background BLACK)\n    # param 5: Block Size (Size of the neighborhood to look at. Must be odd number, e.g., 11, 21, 51)\n    # param 6: C (Constant subtracted from the mean. Fine-tunes noise sensitivity)\n    adaptive_thresh = cv2.adaptiveThreshold(\n        gray, \n        255, \n        cv2.ADAPTIVE_THRESH_GAUSSIAN_C, \n        cv2.THRESH_BINARY_INV, \n        51,  # Large block size helps ignore the grid lines but keep the trace\n        10   # Constant\n    )\n\n    # 5. Visualization\n    plt.figure(figsize=(20, 10))\n\n    # Original\n    plt.subplot(1, 3, 1)\n    plt.imshow(cv2.cvtColor(original, cv2.COLOR_BGR2RGB))\n    plt.title(\"Original Photo (With Shadows)\")\n    plt.axis('off')\n\n    # Global Failure\n    plt.subplot(1, 3, 2)\n    plt.imshow(global_thresh, cmap='gray')\n    plt.title(\"Global Threshold (Shadow becomes Signal!)\")\n    plt.axis('off')\n\n    # Adaptive Success\n    plt.subplot(1, 3, 3)\n    plt.imshow(adaptive_thresh, cmap='gray')\n    plt.title(\"Adaptive Threshold (Ignores Shadow)\")\n    plt.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\nsolve_shadows_with_adaptive_threshold()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:10:15.699959Z","iopub.execute_input":"2025-12-27T07:10:15.700275Z","iopub.status.idle":"2025-12-27T07:10:25.641498Z","shell.execute_reply.started":"2025-12-27T07:10:15.700224Z","shell.execute_reply":"2025-12-27T07:10:25.640318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Reuse the image path you defined in the previous step\n# (Or manually set the path to the 'shadow' image again)\nimg_path = '/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939-0001.png' # Update this if needed\n\ndef remove_grid_lines(img_path):\n    # 1. Pre-processing (Repeat previous steps)\n    original = cv2.imread(img_path)\n    if original is None:\n        print(\"Image not found. Check path.\")\n        return\n        \n    gray = cv2.cvtColor(original, cv2.COLOR_BGR2GRAY)\n    \n    # Adaptive Threshold (The step you just proved works)\n    # We use a slightly smaller block size here to keep the signal sharp\n    binary = cv2.adaptiveThreshold(\n        gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, \n        cv2.THRESH_BINARY_INV, 51, 10\n    )\n\n    # 2. Morphological Operations (The New Step)\n    \n    # Define the \"Kernel\" (The shape we check against)\n    # A 3x3 square is usually larger than a grid line (1px) but smaller than the signal\n    kernel_size = 3\n    kernel = np.ones((kernel_size, kernel_size), np.uint8)\n\n    # MORPH_OPEN: Erosion -> Dilation (Removes thin noise)\n    # This eats away the thin grid lines\n    cleaned = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel)\n\n    # MORPH_CLOSE: Dilation -> Erosion (Fills small gaps)\n    # This helps reconnect parts of the signal that might have broken\n    cleaned = cv2.morphologyEx(cleaned, cv2.MORPH_CLOSE, kernel)\n\n    # 3. Visualization\n    plt.figure(figsize=(18, 10))\n\n    # Input (Noisy Adaptive Threshold)\n    plt.subplot(1, 2, 1)\n    plt.imshow(binary, cmap='gray')\n    plt.title(\"Step 1: Adaptive Threshold (Signal + Grid Noise)\")\n    plt.axis('off')\n\n    # Output (Cleaned)\n    plt.subplot(1, 2, 2)\n    plt.imshow(cleaned, cmap='gray')\n    plt.title(\"Step 2: Grid Removed (Morphological Opening)\")\n    plt.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\n# Run it (Make sure to point to the correct image file!)\n# You can use the glob method from before to find the ID '1015663939' again if needed\nimport glob\nfiles = glob.glob(f'/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939-0001.png', recursive=True)\nif files:\n    remove_grid_lines(files[0])\nelse:\n    print(\"Could not find the monitor image. Please check the ID or path.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:10:25.643711Z","iopub.execute_input":"2025-12-27T07:10:25.644525Z","iopub.status.idle":"2025-12-27T07:10:26.739367Z","shell.execute_reply.started":"2025-12-27T07:10:25.644478Z","shell.execute_reply":"2025-12-27T07:10:26.738280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport glob\n\ndef custom_crop_ecg(img_path, top_trim_percent=0.15, left_padding_percent=0.03, right_padding_percent=0.03):\n    \"\"\"\n    top_trim_percent: Cuts off the top (0.15 = 15%).\n    left_padding_percent: Adds margin to the LEFT.\n    right_padding_percent: Adds margin to the RIGHT.\n    \"\"\"\n    if not img_path:\n        print(\"Image path is None.\")\n        return None\n\n    original = cv2.imread(img_path)\n    if original is None: return None\n    h_img, w_img = original.shape[:2]\n\n    # 1. Isolate Signal\n    gray = cv2.cvtColor(original, cv2.COLOR_BGR2GRAY)\n    _, binary = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n    kernel = np.ones((3,3), np.uint8)\n    binary = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel)\n\n    # 2. Find Initial Bounding Box\n    points = cv2.findNonZero(binary)\n    if points is None: return original\n    x, y, w, h = cv2.boundingRect(points)\n\n    # --- THE ADJUSTMENTS ---\n    \n    # 1. Top Trim (Moving Down)\n    pixels_to_shift_down = int(h * top_trim_percent)\n    y_new = y + pixels_to_shift_down\n    h_new = h - pixels_to_shift_down\n    \n    # 2. Left Padding (Moving Left)\n    pixels_to_shift_left = int(w * left_padding_percent)\n    x_new = max(0, x - pixels_to_shift_left)\n\n    # 3. Right Padding (Moving Right) <--- NEW\n    pixels_to_shift_right = int(w * right_padding_percent)\n    # Calculate the new end point (x + w + extra), ensuring we don't go past image width\n    x_end = min(w_img, x + w + pixels_to_shift_right)\n\n    # Safety check for height\n    if h_new <= 0:\n        print(\"Warning: Top trim value too high! Resetting top.\")\n        y_new, h_new = y, h\n\n    # 3. Final Crop\n    # Slice y: from new start to new height\n    # Slice x: from new left (x_new) to new right (x_end)\n    cropped_img = original[y_new : y_new+h_new, x_new : x_end]\n\n    # --- Visualization ---\n    plt.figure(figsize=(12, 6))\n    \n    plt.subplot(1, 2, 1)\n    debug_img = original.copy()\n    # Green Box = Original detection\n    cv2.rectangle(debug_img, (x, y), (x+w, y+h), (0, 255, 0), 2)\n    # Red Box = Adjusted crop\n    cv2.rectangle(debug_img, (x_new, y_new), (x_end, y_new+h_new), (0, 0, 255), 4)\n    \n    plt.imshow(cv2.cvtColor(debug_img, cv2.COLOR_BGR2RGB))\n    plt.title(f\"Green=Original, Red=Final\\n(L-Pad: {left_padding_percent}, R-Pad: {right_padding_percent})\")\n    plt.axis('off')\n\n    plt.subplot(1, 2, 2)\n    plt.imshow(cv2.cvtColor(cropped_img, cv2.COLOR_BGR2RGB))\n    plt.title(\"Final Result\")\n    plt.axis('off')\n    \n    plt.show()\n\n    return cropped_img\n\n# --- RUN IT ---\nfiles = glob.glob(f'/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939-0001.png', recursive=True)\nimg_path = files[0] if files else None\n\nprint(f\"Processing: {img_path}\")\n# Adjust 'right_padding_percent' here (e.g., 0.05 for 5% more width)\nfinal_crop = custom_crop_ecg(img_path, top_trim_percent=0.3, left_padding_percent=0.05, right_padding_percent=0.05)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:10:49.128689Z","iopub.execute_input":"2025-12-27T07:10:49.129635Z","iopub.status.idle":"2025-12-27T07:10:50.182812Z","shell.execute_reply.started":"2025-12-27T07:10:49.129604Z","shell.execute_reply":"2025-12-27T07:10:50.181777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\n# Use the 'final_crop' from your previous step\ndef segment_leads_with_color_trick(image):\n    if image is None:\n        print(\"No image provided.\")\n        return\n\n    # 1. The Red Channel Trick\n    # OpenCV loads images as BGR (Blue, Green, Red).\n    # Channel 0 = Blue, 1 = Green, 2 = Red.\n    # We grab Channel 2. The pink grid (high Red value) becomes bright/white.\n    red_channel = image[:, :, 2] \n    \n    # 2. Thresholding (Inverted)\n    # We want the Black signal to be White (1) and the White background to be Black (0).\n    # Since the grid is now \"White\" in the Red channel, it gets turned to Black (Background).\n    _, binary = cv2.threshold(red_channel, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n\n    # 3. Horizontal Projection (Same as before)\n    projection = np.sum(binary, axis=1)\n\n    # Smooth the graph to make the \"mountains\" clearer\n    kernel_size = 50\n    smooth_projection = np.convolve(projection, np.ones(kernel_size)/kernel_size, mode='same')\n\n    # 4. Find the Valleys (The Cut Points)\n    # We look for areas where the signal density drops significantly\n    # A simple way is to find the minimum points between the peaks.\n    # For this demo, we'll divide the image into 4 equal chunks if peaks aren't clear,\n    # or visual the improved projection first.\n    \n    plt.figure(figsize=(15, 8))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(smooth_projection, range(len(smooth_projection)))\n    plt.gca().invert_yaxis()\n    plt.title(\"Projection (Using Red Channel)\")\n    plt.xlabel(\"Pixel Density\")\n    plt.ylabel(\"Row Number\")\n    plt.grid(True, alpha=0.3)\n\n    plt.subplot(1, 2, 2)\n    plt.imshow(binary, cmap='gray')\n    plt.title(\"Binary Map (Grid Should Be Gone)\")\n    plt.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\n# Run it\nif 'final_crop' in locals():\n    segment_leads_with_color_trick(final_crop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:11:01.573742Z","iopub.execute_input":"2025-12-27T07:11:01.574029Z","iopub.status.idle":"2025-12-27T07:11:02.105917Z","shell.execute_reply.started":"2025-12-27T07:11:01.574007Z","shell.execute_reply":"2025-12-27T07:11:02.105182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef slice_ecg_rows(image, projection):\n    \"\"\"\n    Slices the ECG image into separate rows based on the projection valleys.\n    \"\"\"\n    # 1. Find the split points (Minima in projection)\n    # We look for the lowest points in the projection to decide where to cut.\n    # We split the image into 4 roughly equal sections first to find the local minima.\n    h, w = image.shape[:2]\n    \n    # Define approximate zones for the 4 cuts (0%, 25%, 50%, 75%, 100% height)\n    # The 'valleys' should be roughly at 25%, 50%, and 75% of the image height.\n    split_zones = [\n        (int(h*0.20), int(h*0.30)), # Gap between Row 1 & 2\n        (int(h*0.45), int(h*0.55)), # Gap between Row 2 & 3\n        (int(h*0.70), int(h*0.80))  # Gap between Row 3 & 4\n    ]\n    \n    cut_indices = [0] # Start at top\n    \n    for start, end in split_zones:\n        # Find the row with the MINIMUM pixel density in this zone\n        # This is the \"emptiest\" horizontal line, perfect for cutting.\n        local_min_idx = np.argmin(projection[start:end]) + start\n        cut_indices.append(local_min_idx)\n        \n    cut_indices.append(h) # End at bottom\n\n    # 2. Slice the Image\n    row_images = []\n    for i in range(len(cut_indices) - 1):\n        y1 = cut_indices[i]\n        y2 = cut_indices[i+1]\n        \n        # Extract the strip\n        strip = image[y1:y2, :]\n        row_images.append(strip)\n\n    # 3. Visualization\n    plt.figure(figsize=(15, 12))\n    for i, strip in enumerate(row_images):\n        plt.subplot(4, 1, i+1)\n        plt.imshow(strip, cmap='gray') # Assuming binary input, or use cv2.cvtColor for color\n        plt.title(f\"Extracted Row {i+1}\")\n        plt.axis('off')\n        \n        # Add a colored border to prove it's a separate image\n        plt.gca().add_patch(plt.Rectangle((0,0), strip.shape[1], strip.shape[0], \n                                          edgecolor='red', facecolor='none', lw=2))\n\n    plt.tight_layout()\n    plt.show()\n    \n    return row_images\n\n# --- RUN IT ---\n# We need the 'binary' image and the 'smooth_projection' from your previous step.\n# Assuming you have 'final_crop' from the previous step:\n\n# 1. Re-generate the binary map and projection (copying logic from your successful run)\ngray = cv2.cvtColor(final_crop, cv2.COLOR_BGR2GRAY)\n# Red channel extraction is often better for removing red grids, but let's stick to what worked for you:\n_, binary_map = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV) \n\n# Calculate Projection again\nprojection = np.sum(binary_map, axis=1)\nkernel_size = 50\nsmooth_projection = np.convolve(projection, np.ones(kernel_size)/kernel_size, mode='same')\n\n# 2. Slice!\necg_rows = slice_ecg_rows(final_crop, smooth_projection)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:12:08.798608Z","iopub.execute_input":"2025-12-27T07:12:08.799395Z","iopub.status.idle":"2025-12-27T07:12:09.733026Z","shell.execute_reply.started":"2025-12-27T07:12:08.799366Z","shell.execute_reply":"2025-12-27T07:12:09.732084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef segment_leads_strict(row_images):\n    if not row_images:\n        print(\"No rows provided!\")\n        return {}\n\n    # Standard 12-Lead Names (3 Rows x 4 Columns)\n    # Row 1: I, aVR, V1, V4\n    # Row 2: II, aVL, V2, V5\n    # Row 3: III, aVF, V3, V6\n    lead_names_grid = [\n        [\"I\", \"aVR\", \"V1\", \"V4\"],\n        [\"II\", \"aVL\", \"V2\", \"V5\"],\n        [\"III\", \"aVF\", \"V3\", \"V6\"]\n    ]\n    \n    final_leads = {}\n    \n    # Setup Visualization\n    plt.figure(figsize=(20, 10))\n    plot_idx = 1\n    \n    # We only process the top 3 rows for the grid\n    # (Row 4 is the Rhythm strip and is handled separately)\n    for row_idx in range(min(3, len(row_images))):\n        current_row = row_images[row_idx]\n        h, w = current_row.shape[:2]\n        \n        # --- THE FIX: STRICT MATHEMATICAL SLICING ---\n        # Instead of searching for gaps, we cut exactly at 1/4, 1/2, and 3/4.\n        # This works because we already cropped the image edges in Step 1.\n        col_width = int(w / 4)\n        \n        cuts = [\n            0,              # Start\n            col_width,      # 25%\n            col_width * 2,  # 50%\n            col_width * 3,  # 75%\n            w               # End\n        ]\n        \n        for col_idx in range(4):\n            x1 = cuts[col_idx]\n            x2 = cuts[col_idx+1]\n            \n            # Slice\n            lead_img = current_row[:, x1:x2]\n            \n            # Store\n            lead_name = lead_names_grid[row_idx][col_idx]\n            final_leads[lead_name] = lead_img\n            \n            # Visualization\n            plt.subplot(3, 4, plot_idx)\n            plt.imshow(lead_img, cmap=\"gray\")\n            plt.title(f\"Lead {lead_name}\")\n            plt.axis('off')\n            \n            # Draw a clean border to verify the content\n            plt.gca().add_patch(plt.Rectangle((0,0), x2-x1, h, \n                                            edgecolor='cyan', facecolor='none', lw=2))\n            plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n\n    # Handle Rhythm Strip (Row 4)\n    if len(row_images) > 3:\n        # The Rhythm strip is usually the full width\n        rhythm_img = row_images[3]\n        final_leads['Rhythm'] = rhythm_img\n        \n        print(\"Extracted Rhythm Strip (Row 4)\")\n        plt.figure(figsize=(20, 3))\n        plt.imshow(rhythm_img, cmap='gray')\n        plt.title(\"Rhythm Strip\")\n        plt.axis('off')\n        plt.show()\n        \n    return final_leads\n\n# --- RUN IT ---\n# Ensure you have 'ecg_rows' from the previous step!\nif 'ecg_rows' in locals():\n    leads = segment_leads_strict(ecg_rows)\nelse:\n    print(\"Error: 'ecg_rows' variable missing. Please run the Horizontal Slicing step first.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:15:55.975467Z","iopub.execute_input":"2025-12-27T07:15:55.975803Z","iopub.status.idle":"2025-12-27T07:15:57.564615Z","shell.execute_reply.started":"2025-12-27T07:15:55.975779Z","shell.execute_reply":"2025-12-27T07:15:57.563751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef segment_leads_with_markers(row_images):\n    if not row_images:\n        print(\"No rows provided!\")\n        return {}\n\n    lead_names_grid = [\n        [\"I\", \"aVR\", \"V1\", \"V4\"],\n        [\"II\", \"aVL\", \"V2\", \"V5\"],\n        [\"III\", \"aVF\", \"V3\", \"V6\"]\n    ]\n    \n    final_leads = {}\n    \n    plt.figure(figsize=(20, 12))\n    plot_idx = 1\n    \n    for row_idx in range(min(3, len(row_images))):\n        current_row = row_images[row_idx]\n        h, w = current_row.shape[:2]\n        \n        # --- STEP 1: Detect Vertical Bars (The Markers) ---\n        # 1. Convert to binary (inverted) if not already\n        if len(current_row.shape) == 3:\n            gray = cv2.cvtColor(current_row, cv2.COLOR_BGR2GRAY)\n            _, binary = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n        else:\n            binary = current_row.copy()\n\n        # 2. ISOLATE VERTICAL LINES\n        # We use a morphological kernel that is tall and thin (1 pixel wide, 20 pixels tall).\n        # This erases the ECG waves (which are horizontal/diagonal) and text, leaving ONLY vertical bars.\n        vertical_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, 20))\n        vertical_lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, vertical_kernel)\n        \n        # 3. Project to 1D (Sum columns)\n        # Any column with a vertical bar will have a HUGE sum.\n        vertical_projection = np.sum(vertical_lines, axis=0)\n        \n        # --- STEP 2: Smart \"Snap-to-Marker\" Logic ---\n        # We expect markers roughly at 0, 25%, 50%, 75% of width\n        expected_locs = [0, w//4, w//2, (w*3)//4]\n        found_markers = []\n        \n        search_window = int(w * 0.05) # Search +/- 5% around the expected area\n        \n        for expected in expected_locs:\n            # Define search bounds\n            start_search = max(0, expected - search_window)\n            end_search = min(w, expected + search_window)\n            \n            # Find the distinct PEAK in this region\n            region = vertical_projection[start_search:end_search]\n            \n            if np.max(region) > (255 * 10): # Threshold: Must be at least 10 pixels high to count\n                # Found a bar!\n                local_max_idx = np.argmax(region)\n                actual_marker = start_search + local_max_idx\n                found_markers.append(actual_marker)\n            else:\n                # No bar found? Fallback to the strict math\n                found_markers.append(expected)\n        \n        # Add the end of the image as the final boundary\n        found_markers.append(w)\n        \n        # --- STEP 3: Slice & Visualize ---\n        for col_idx in range(4):\n            x1 = found_markers[col_idx]\n            x2 = found_markers[col_idx+1]\n            \n            # Safety check: Ensure x2 > x1 (markers aren't on top of each other)\n            if x2 - x1 < 50: x2 = x1 + (w//4) # Fallback if detection failed completely\n            \n            # Slice\n            lead_img = current_row[:, x1:x2]\n            \n            lead_name = lead_names_grid[row_idx][col_idx]\n            final_leads[lead_name] = lead_img\n            \n            # Visualization\n            plt.subplot(3, 4, plot_idx)\n            plt.imshow(lead_img, cmap=\"gray\")\n            plt.title(f\"{lead_name}\")\n            plt.axis('off')\n            \n            # Visual Debug: Show where we cut\n            # Green Line = The Start Marker we found\n            # Blue Box = The Lead content\n            plt.gca().add_patch(plt.Rectangle((0,0), x2-x1, h, edgecolor='blue', facecolor='none', lw=1))\n            # Draw the marker explicitly to show user we found it\n            plt.axvline(x=0, color='r', linewidth=5, alpha=0.5) \n            \n            plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n    \n    # Handle Rhythm Strip\n    if len(row_images) > 3:\n        final_leads['Rhythm'] = row_images[3]\n\n    return final_leads\n\n# --- RUN IT ---\nif 'ecg_rows' in locals():\n    # This should now perfectly align with the '|' bars\n    final_leads = segment_leads_with_markers(ecg_rows)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:20:48.144561Z","iopub.execute_input":"2025-12-27T07:20:48.145255Z","iopub.status.idle":"2025-12-27T07:20:49.604205Z","shell.execute_reply.started":"2025-12-27T07:20:48.145198Z","shell.execute_reply":"2025-12-27T07:20:49.603000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef segment_leads_global_align(row_images):\n    if not row_images:\n        print(\"No rows provided!\")\n        return {}\n\n    lead_names_grid = [\n        [\"I\", \"aVR\", \"V1\", \"V4\"],\n        [\"II\", \"aVL\", \"V2\", \"V5\"],\n        [\"III\", \"aVF\", \"V3\", \"V6\"]\n    ]\n    \n    final_leads = {}\n    \n    # 1. Combine Top 3 Rows to find Global Markers\n    # We stack them vertically to make the vertical lines continuous and strong\n    top_3_rows = []\n    for i in range(min(3, len(row_images))):\n        img = row_images[i]\n        # Ensure binary\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            _, bin_img = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n        else:\n            bin_img = img.copy()\n        top_3_rows.append(bin_img)\n    \n    # Create the \"Master Projection\"\n    combined_img = np.vstack(top_3_rows)\n    h_total, w = combined_img.shape\n    \n    # 2. Detect Vertical Lines on the Master Image\n    # Morphological operator to isolate ONLY vertical lines\n    vertical_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, 30)) # Tall kernel\n    vertical_lines = cv2.morphologyEx(combined_img, cv2.MORPH_OPEN, vertical_kernel)\n    \n    # Sum columns to find the peaks\n    global_projection = np.sum(vertical_lines, axis=0)\n    \n    # 3. Find the 3 Global Cut Points (approx at 25%, 50%, 75%)\n    # We assume the first column starts at 0. We need to find the start of Col 2, 3, and 4.\n    expected_locs = [int(w*0.25), int(w*0.50), int(w*0.75)]\n    found_cuts = [0] # Start of first col is always 0\n    \n    search_window = int(w * 0.05) # +/- 5% wiggle room\n    \n    print(\"Detected Global Cut X-Coordinates:\")\n    for expected in expected_locs:\n        # Search window around expected location\n        start_s = max(0, expected - search_window)\n        end_s = min(w, expected + search_window)\n        \n        region = global_projection[start_s:end_s]\n        \n        # If we find a strong peak (the vertical bar)\n        if np.max(region) > (255 * 20): \n            local_max = np.argmax(region)\n            cut_x = start_s + local_max\n            print(f\" - Found marker at x={cut_x} (Expected ~{expected})\")\n        else:\n            # Fallback to strict math if the line is missing/faint\n            print(f\" - No marker found near {expected}. Using default.\")\n            cut_x = expected\n            \n        found_cuts.append(cut_x)\n    \n    found_cuts.append(w) # End of image\n    \n    # 4. Apply these GLOBAL cuts to specific rows\n    plt.figure(figsize=(20, 12))\n    plot_idx = 1\n    \n    for row_idx in range(min(3, len(row_images))):\n        current_row = row_images[row_idx]\n        \n        for col_idx in range(4):\n            x1 = found_cuts[col_idx]\n            x2 = found_cuts[col_idx+1]\n            \n            # Slice\n            lead_img = current_row[:, x1:x2]\n            \n            # Store\n            lead_name = lead_names_grid[row_idx][col_idx]\n            final_leads[lead_name] = lead_img\n            \n            # Visualize\n            plt.subplot(3, 4, plot_idx)\n            plt.imshow(lead_img, cmap=\"gray\")\n            plt.title(f\"{lead_name}\")\n            plt.axis('off')\n            # Draw blue box to show the cut\n            plt.gca().add_patch(plt.Rectangle((0,0), x2-x1, current_row.shape[0], \n                                            edgecolor='blue', facecolor='none', lw=2))\n            \n            plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n    \n    # 5. Handle Rhythm Strip (Row 4)\n    # The user mentioned \"missed the down one\". This handles the bottom strip.\n    if len(row_images) > 3:\n        rhythm_img = row_images[3]\n        final_leads['Rhythm'] = rhythm_img\n        \n        print(\"Processing Rhythm Strip (Row 4)...\")\n        plt.figure(figsize=(20, 3))\n        plt.imshow(rhythm_img, cmap='gray')\n        plt.title(\"Rhythm Strip (Continuous Lead II)\")\n        plt.axis('off')\n        plt.show()\n\n    return final_leads\n\n# --- RUN IT ---\nif 'ecg_rows' in locals():\n    # This aligns ALL rows to the same vertical markers\n    final_leads = segment_leads_global_align(ecg_rows)\nelse:\n    print(\"Error: 'ecg_rows' missing.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:24:49.976168Z","iopub.execute_input":"2025-12-27T07:24:49.976591Z","iopub.status.idle":"2025-12-27T07:24:51.791123Z","shell.execute_reply.started":"2025-12-27T07:24:49.976563Z","shell.execute_reply":"2025-12-27T07:24:51.790186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef segment_leads_with_shift(row_images, shift_col_4=0):\n    \"\"\"\n    Segments the ECG into 12 distinct leads + 1 Rhythm strip.\n    \n    Parameters:\n    - row_images: List of 4 image strips (Top 3 are grid, 4th is Rhythm).\n    - shift_col_4: Manual pixel shift for the start of the V4/V5/V6 column.\n                   Positive = Move Right, Negative = Move Left.\n    \"\"\"\n    if not row_images:\n        print(\"No rows provided!\")\n        return {}\n\n    # Standard 12-Lead Layout Names (3 Rows x 4 Columns)\n    lead_names_grid = [\n        [\"I\", \"aVR\", \"V1\", \"V4\"],\n        [\"II\", \"aVL\", \"V2\", \"V5\"],\n        [\"III\", \"aVF\", \"V3\", \"V6\"]\n    ]\n    \n    final_leads = {}\n    \n    # --- STEP 1: GLOBAL VERTICAL ALIGNMENT ---\n    # We combine the top 3 rows to find the strong vertical dividers\n    top_3_rows = []\n    # Only loop through the first 3 rows (ignoring the rhythm strip for now)\n    for i in range(min(3, len(row_images))):\n        img = row_images[i]\n        # Ensure we are working with binary images\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            _, bin_img = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n        else:\n            bin_img = img.copy()\n        top_3_rows.append(bin_img)\n    \n    # Stack them to create a \"Master Image\" for detection\n    combined_img = np.vstack(top_3_rows)\n    h_total, w = combined_img.shape\n    \n    # Detect Vertical Lines using morphology\n    vertical_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, 30))\n    vertical_lines = cv2.morphologyEx(combined_img, cv2.MORPH_OPEN, vertical_kernel)\n    global_projection = np.sum(vertical_lines, axis=0)\n    \n    # Expected locations for cuts: 25%, 50%, 75% of width\n    expected_locs = [int(w*0.25), int(w*0.50), int(w*0.75)]\n    found_cuts = [0] # Column 1 always starts at 0\n    \n    search_window = int(w * 0.05) # +/- 5% search range\n    \n    print(\"--- CUTTING CALCULATIONS ---\")\n    for i, expected in enumerate(expected_locs):\n        # 1. Search for the strongest vertical line in the area\n        start_s = max(0, expected - search_window)\n        end_s = min(w, expected + search_window)\n        region = global_projection[start_s:end_s]\n        \n        if np.max(region) > (255 * 20): # Threshold: Real line found\n            local_max = np.argmax(region)\n            cut_x = start_s + local_max\n        else:\n            cut_x = expected # Fallback to math\n            \n        # 2. Apply MANUAL SHIFT for the 3rd cut (V4/V5/V6)\n        # This is the line that separates V1/V2/V3 from V4/V5/V6\n        if i == 2: \n            original_cut = cut_x\n            cut_x += shift_col_4\n            print(f\"Cut {i+1} (V4 start): Found at {original_cut}, Shifted to {cut_x} (Delta: {shift_col_4})\")\n        else:\n            print(f\"Cut {i+1}: Found at {cut_x}\")\n            \n        found_cuts.append(cut_x)\n    \n    found_cuts.append(w) # End of image\n    \n    # --- STEP 2: SLICE AND STORE (TOP 3 ROWS) ---\n    plt.figure(figsize=(20, 12))\n    plot_idx = 1\n    \n    for row_idx in range(min(3, len(row_images))):\n        current_row = row_images[row_idx]\n        \n        for col_idx in range(4):\n            x1 = found_cuts[col_idx]\n            x2 = found_cuts[col_idx+1]\n            \n            # Slice the lead\n            lead_img = current_row[:, x1:x2]\n            \n            # Store it\n            lead_name = lead_names_grid[row_idx][col_idx]\n            final_leads[lead_name] = lead_img\n            \n            # Visualization\n            plt.subplot(4, 4, plot_idx) # 4 rows now (including rhythm)\n            plt.imshow(lead_img, cmap=\"gray\")\n            plt.title(f\"{lead_name}\")\n            plt.axis('off')\n            \n            # Visual check: Blue box around content, Red line at V4 start\n            if col_idx == 3:\n                # Mark the V4/V5/V6 start explicitly\n                plt.axvline(x=0, color='blue', linewidth=1, alpha=0.7)\n            else:\n                plt.gca().add_patch(plt.Rectangle((0,0), x2-x1, current_row.shape[0], edgecolor='blue', facecolor='none', lw=1))\n            \n            plot_idx += 1\n\n    # --- STEP 3: RHYTHM STRIP (ROW 4) ---\n    if len(row_images) > 3:\n        rhythm_img = row_images[3]\n        final_leads['II_long'] = rhythm_img # Standard name for Rhythm strip\n        \n        # Display it at the bottom, spanning full width\n        # We use a new subplot that spans the entire bottom row\n        plt.subplot(4, 1, 4) \n        plt.imshow(rhythm_img, cmap='gray')\n        plt.title(\"Lead II (Long Rhythm Strip)\")\n        plt.axis('off')\n        \n    plt.tight_layout()\n    plt.show()\n\n    return final_leads\n\n# --- EXECUTION ---\nif 'ecg_rows' in locals():\n    print(\"Segmenting with shift_col_4 = 17...\")\n    final_leads = segment_leads_with_shift(ecg_rows, shift_col_4=17)\n    \n    print(\"\\nExtraction Complete.\")\n    print(f\"Extracted: {list(final_leads.keys())}\")\nelse:\n    print(\"Error: 'ecg_rows' variable missing. Please run the Horizontal Slicing step first.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:40:52.120074Z","iopub.execute_input":"2025-12-27T07:40:52.120465Z","iopub.status.idle":"2025-12-27T07:40:53.779791Z","shell.execute_reply.started":"2025-12-27T07:40:52.120436Z","shell.execute_reply":"2025-12-27T07:40:53.778868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\ndef extract_1d_signals(leads_dict):\n    \"\"\"\n    Converts lead images into 1D signal arrays (Voltage vs Time).\n    \"\"\"\n    signals_1d = {}\n    \n    plt.figure(figsize=(20, 12))\n    plot_idx = 1\n    \n    # Sort keys to ensure standard order for plotting\n    standard_order = [\n        \"I\", \"II\", \"III\", \n        \"aVR\", \"aVL\", \"aVF\", \n        \"V1\", \"V2\", \"V3\", \n        \"V4\", \"V5\", \"V6\"\n    ]\n    \n    # Add Rhythm if present\n    if \"II_long\" in leads_dict:\n        standard_order.append(\"II_long\")\n\n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n            \n        img = leads_dict[lead_name]\n        \n        # 1. Preprocessing\n        # Ensure binary (0 = Black/Signal, 255 = White/Background)\n        # We invert it so Signal is White (1) and Background is Black (0) for easier calculation\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else:\n            gray = img\n            \n        # Threshold: Signal becomes 1, Background becomes 0\n        _, binary = cv2.threshold(gray, 120, 1, cv2.THRESH_BINARY_INV)\n        \n        # 2. Extract Signal (Column by Column)\n        h, w = binary.shape\n        signal = []\n        \n        for col in range(w):\n            # Get all pixels in this column\n            column_pixels = binary[:, col]\n            \n            # Find indices where pixel is \"Signal\" (Value 1)\n            signal_indices = np.where(column_pixels > 0)[0]\n            \n            if len(signal_indices) > 0:\n                # Calculate the median Y-position of the signal line thickness\n                y_center = np.median(signal_indices)\n            else:\n                # If no signal found in this column (gap), use the previous value or NaN\n                y_center = signal[-1] if len(signal) > 0 else h // 2\n            \n            # 3. Invert Y-Axis\n            # In images, 0 is at the top. In graphs, positive is up.\n            # So: Value = (Height - y_center)\n            signal.append(h - y_center)\n            \n        # Convert to numpy array\n        signal = np.array(signal)\n        \n        # 4. Zero-Centering (Remove DC Offset)\n        # We assume the \"Isoelectric Line\" (0 mV) is roughly the median of the signal\n        signal_centered = signal - np.median(signal)\n        \n        signals_1d[lead_name] = signal_centered\n        \n        # 5. Visualization\n        plt.subplot(4, 4, plot_idx)\n        if lead_name == \"II_long\":\n            plt.subplot(4, 1, 4) # Span bottom row\n            \n        plt.plot(signal_centered, color='black', linewidth=1)\n        plt.title(f\"Digitized {lead_name}\")\n        plt.grid(True, alpha=0.3)\n        plt.xlim(0, len(signal_centered))\n        plt.axis('off') # Hide axis numbers for cleaner look\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n    \n    return signals_1d\n\n# --- RUN IT ---\n# This converts your images into raw data arrays!\nif 'final_leads' in locals():\n    ecg_data = extract_1d_signals(final_leads)\n    \n    # Example: Check the shape of Lead I\n    print(f\"Lead I extracted. Length: {len(ecg_data['I'])} samples.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:43:54.611405Z","iopub.execute_input":"2025-12-27T07:43:54.611781Z","iopub.status.idle":"2025-12-27T07:43:55.933136Z","shell.execute_reply.started":"2025-12-27T07:43:54.611756Z","shell.execute_reply":"2025-12-27T07:43:55.932286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef overlay_extraction_on_images(leads_dict, signals_dict):\n    \"\"\"\n    Draws the extracted 1D signal (Green Line) over the original image\n    to verify accuracy.\n    \"\"\"\n    if not leads_dict or not signals_dict:\n        print(\"Missing data to visualize!\")\n        return\n\n    # Create a list for plotting\n    lead_names = [\n        \"I\", \"II\", \"III\", \n        \"aVR\", \"aVL\", \"aVF\", \n        \"V1\", \"V2\", \"V3\", \n        \"V4\", \"V5\", \"V6\"\n    ]\n    if \"II_long\" in leads_dict: lead_names.append(\"II_long\")\n    \n    plt.figure(figsize=(20, 15))\n    plot_idx = 1\n    \n    for name in lead_names:\n        if name not in leads_dict: continue\n            \n        # 1. Get Image and Signal\n        original_img = leads_dict[name].copy()\n        signal = signals_dict[name]\n        \n        # Convert to Color (so we can draw a Green line)\n        if len(original_img.shape) == 2:\n            color_img = cv2.cvtColor(original_img, cv2.COLOR_GRAY2BGR)\n        else:\n            color_img = original_img\n\n        h, w = color_img.shape[:2]\n        \n        # 2. Map Signal back to Image Coordinates\n        # Recall: signal_centered = (h - y) - median\n        # Therefore: y = h - median - signal_centered\n        # We approximate 'median' as h/2 for visualization\n        points = []\n        for x in range(min(w, len(signal))):\n            val = signal[x]\n            \n            # Un-center the signal to find the pixel coordinate\n            # (h/2 is the approximate center line)\n            y = int((h / 2) - val) \n            \n            # Clip to ensure we don't draw outside image\n            y = max(0, min(h - 1, y))\n            \n            points.append((x, y))\n            \n        # 3. Draw the Line\n        # Convert list of points to numpy array for polylines\n        pts = np.array(points, np.int32)\n        pts = pts.reshape((-1, 1, 2))\n        \n        # Draw GREEN line (BGR: 0, 255, 0), thickness 2\n        cv2.polylines(color_img, [pts], isClosed=False, color=(0, 255, 0), thickness=2)\n        \n        # 4. Show it\n        plt.subplot(4, 4, plot_idx)\n        if name == \"II_long\": plt.subplot(4, 1, 4)\n            \n        plt.imshow(cv2.cvtColor(color_img, cv2.COLOR_BGR2RGB))\n        plt.title(f\"{name} (Green = Extracted Data)\")\n        plt.axis('off')\n        \n        plot_idx += 1\n        \n    plt.tight_layout()\n    plt.show()\n\n# --- RUN IT ---\nif 'final_leads' in locals() and 'ecg_data' in locals():\n    overlay_extraction_on_images(final_leads, ecg_data)\nelse:\n    print(\"Error: Missing 'final_leads' or 'ecg_data'. Run previous steps first.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:45:52.882077Z","iopub.execute_input":"2025-12-27T07:45:52.882472Z","iopub.status.idle":"2025-12-27T07:45:54.711495Z","shell.execute_reply.started":"2025-12-27T07:45:52.882446Z","shell.execute_reply":"2025-12-27T07:45:54.710605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef extract_1d_signals_clean(leads_dict):\n    \"\"\"\n    Converts lead images into 1D signal arrays, REMOVING TEXT NOISE.\n    \"\"\"\n    signals_1d = {}\n    \n    # Sort keys for plotting order\n    standard_order = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    if \"II_long\" in leads_dict: standard_order.append(\"II_long\")\n\n    plt.figure(figsize=(20, 15))\n    plot_idx = 1\n    \n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n            \n        img = leads_dict[lead_name]\n        \n        # 1. Preprocessing (Isolate Black Pixels)\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else:\n            gray = img\n            \n        # Threshold: Signal is White (255), Background is Black (0)\n        _, binary = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n        \n        # --- THE FIX: REMOVE TEXT (NOISE CLEANING) ---\n        # Find all connected blobs (components)\n        num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(binary, connectivity=8)\n        \n        # Create a blank mask to draw ONLY the \"real\" signal\n        clean_mask = np.zeros_like(binary)\n        \n        # Loop through all found components (skip label 0, which is background)\n        for i in range(1, num_labels):\n            width = stats[i, cv2.CC_STAT_WIDTH]\n            height = stats[i, cv2.CC_STAT_HEIGHT]\n            area = stats[i, cv2.CC_STAT_AREA]\n            \n            # LOGIC: \n            # ECG signals are WIDE (spanning across the image).\n            # Text letters are usually small squares.\n            # We keep only components that are significantly wide (> 10% of image width)\n            # OR have a very large area.\n            if width > (binary.shape[1] * 0.15): \n                clean_mask[labels == i] = 255\n                \n        # ---------------------------------------------\n        \n        # 2. Extract Signal from the CLEAN MASK\n        h, w = clean_mask.shape\n        signal = []\n        \n        for col in range(w):\n            column_pixels = clean_mask[:, col]\n            signal_indices = np.where(column_pixels > 0)[0]\n            \n            if len(signal_indices) > 0:\n                y_center = np.median(signal_indices)\n                # Save valid value\n                last_valid = h - y_center\n            else:\n                # Fill gaps (if line breaks) with the last known value\n                last_valid = signal[-1] if len(signal) > 0 else h // 2\n                \n            signal.append(last_valid)\n            \n        signal = np.array(signal)\n        \n        # 3. Zero-Centering (Remove DC Offset)\n        signal_centered = signal - np.median(signal)\n        signals_1d[lead_name] = signal_centered\n        \n        # Visualization\n        plt.subplot(4, 4, plot_idx)\n        if lead_name == \"II_long\": plt.subplot(4, 1, 4)\n            \n        plt.plot(signal_centered, color='green', linewidth=1)\n        plt.title(f\"Cleaned {lead_name}\")\n        plt.grid(True, alpha=0.3)\n        plt.axis('off')\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n    \n    return signals_1d\n\n# --- RUN IT ---\nif 'final_leads' in locals():\n    print(\"Extracting signals with Text Removal...\")\n    ecg_data = extract_1d_signals_clean(final_leads)\n    \n    # Run the overlay check again to see the improvement!\n    overlay_extraction_on_images(final_leads, ecg_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:48:45.429805Z","iopub.execute_input":"2025-12-27T07:48:45.430774Z","iopub.status.idle":"2025-12-27T07:48:48.563470Z","shell.execute_reply.started":"2025-12-27T07:48:45.430742Z","shell.execute_reply":"2025-12-27T07:48:48.562085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef extract_1d_signals_tracker(leads_dict):\n    \"\"\"\n    Digitizes ECG signals using a 'Snake Tracker' algorithm that follows \n    the line continuity to ignore text and isolated noise.\n    \"\"\"\n    signals_1d = {}\n    \n    # Standard plotting order\n    standard_order = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    if \"II_long\" in leads_dict: standard_order.append(\"II_long\")\n\n    plt.figure(figsize=(20, 15))\n    plot_idx = 1\n    \n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n            \n        img = leads_dict[lead_name]\n        \n        # 1. Preprocessing (Aggressive Grid Removal)\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else:\n            gray = img\n            \n        # Threshold: Invert so Signal = 255 (White), Background = 0 (Black)\n        # We use a strict threshold (100) to ensure we miss the faint grid lines\n        _, binary = cv2.threshold(gray, 100, 255, cv2.THRESH_BINARY_INV)\n        \n        # 2. Morphological Cleaning (Connect the Snake)\n        # We use a horizontal kernel. This connects gaps in the ECG line \n        # but does NOT connect the text letters to the line.\n        kernel_h = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 1))\n        binary = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, kernel_h)\n        \n        h, w = binary.shape\n        signal = np.zeros(w)\n        \n        # --- THE SNAKE TRACKER ---\n        \n        # A. Find a starting point (Scan the first 10% of columns)\n        # We look for the most common Y-value to anchor our start.\n        start_candidates = []\n        for col in range(int(w * 0.1)):\n            indices = np.where(binary[:, col] > 0)[0]\n            if len(indices) > 0:\n                start_candidates.extend(indices)\n        \n        if start_candidates:\n            # Start at the median of the beginning pixels\n            current_y = np.median(start_candidates)\n        else:\n            current_y = h // 2 # Fallback to center\n            \n        # B. Walk through the image\n        search_radius = 15 # Look +/- 15 pixels up/down for the next step\n        \n        for col in range(w):\n            # Get all signal pixels in this column\n            col_pixels = np.where(binary[:, col] > 0)[0]\n            \n            if len(col_pixels) > 0:\n                # Filter pixels: Keep only those close to our current path\n                # abs(pixel_y - current_y) < radius\n                candidates = col_pixels[np.abs(col_pixels - current_y) < search_radius]\n                \n                if len(candidates) > 0:\n                    # If we found valid pixels nearby, move to their center\n                    new_y = np.median(candidates)\n                    current_y = new_y # Update state\n                else:\n                    # No pixels nearby? (Maybe a gap or we lost the line)\n                    # HOLD position (don't jump to noise). \n                    # Optionally, expand search radius slightly for next step?\n                    pass \n            \n            # Record the value (Inverted for Graph: Height - Y)\n            signal[col] = h - current_y\n\n        # 3. Post-Processing\n        # Zero-center the signal\n        signal_centered = signal - np.median(signal)\n        signals_1d[lead_name] = signal_centered\n        \n        # Visualization\n        plt.subplot(4, 4, plot_idx)\n        if lead_name == \"II_long\": plt.subplot(4, 1, 4)\n            \n        plt.plot(signal_centered, color='blue', linewidth=1)\n        plt.title(f\"Tracked {lead_name}\")\n        plt.grid(True, alpha=0.3)\n        plt.axis('off')\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n    \n    return signals_1d\n\n# --- RUN IT ---\nif 'final_leads' in locals():\n    print(\"Running Snake Tracker Algorithm...\")\n    ecg_data = extract_1d_signals_tracker(final_leads)\n    \n    # Run the overlay check to confirm the green noise is gone\n    overlay_extraction_on_images(final_leads, ecg_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:53:33.698482Z","iopub.execute_input":"2025-12-27T07:53:33.699310Z","iopub.status.idle":"2025-12-27T07:53:36.763030Z","shell.execute_reply.started":"2025-12-27T07:53:33.699271Z","shell.execute_reply":"2025-12-27T07:53:36.761805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\ndef extract_1d_signals_dynamic(leads_dict):\n    \"\"\"\n    Digitizes ECG signals using a Dynamic Tracker that expands its search \n    window for fast-moving signals (QRS complexes).\n    \"\"\"\n    signals_1d = {}\n    \n    standard_order = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    if \"II_long\" in leads_dict: standard_order.append(\"II_long\")\n\n    plt.figure(figsize=(20, 15))\n    plot_idx = 1\n    \n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n            \n        img = leads_dict[lead_name]\n        \n        # 1. Preprocessing (Same as before)\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else:\n            gray = img\n            \n        _, binary = cv2.threshold(gray, 100, 255, cv2.THRESH_BINARY_INV)\n        kernel_h = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 1))\n        binary = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, kernel_h)\n        \n        h, w = binary.shape\n        signal = np.zeros(w)\n        \n        # --- DYNAMIC TRACKER ---\n        \n        # A. Start Point\n        start_candidates = []\n        for col in range(int(w * 0.05)):\n            indices = np.where(binary[:, col] > 0)[0]\n            if len(indices) > 0:\n                start_candidates.extend(indices)\n        \n        if start_candidates:\n            current_y = np.median(start_candidates)\n        else:\n            current_y = h // 2\n            \n        # B. Walk through image\n        # Base radius is small (to ignore noise), but we allow it to grow.\n        base_radius = 10 \n        max_radius = 80 # Allow huge jumps for QRS\n        \n        for col in range(w):\n            col_pixels = np.where(binary[:, col] > 0)[0]\n            \n            if len(col_pixels) > 0:\n                # 1. Try Small Radius first (Is the line continuing smoothly?)\n                candidates = col_pixels[np.abs(col_pixels - current_y) < base_radius]\n                \n                # 2. If nothing found, Try Large Radius (Did the line spike?)\n                if len(candidates) == 0:\n                    candidates = col_pixels[np.abs(col_pixels - current_y) < max_radius]\n                \n                if len(candidates) > 0:\n                    # Pick the candidate closest to the previous point\n                    # This prevents jumping to a text label that is 70px away if there is a real signal 40px away.\n                    best_candidate = candidates[np.argmin(np.abs(candidates - current_y))]\n                    current_y = best_candidate\n                else:\n                    # Still nothing? Hold position.\n                    pass\n            \n            signal[col] = h - current_y\n\n        # 3. Post-Processing\n        signal_centered = signal - np.median(signal)\n        signals_1d[lead_name] = signal_centered\n        \n        # Visualization\n        plt.subplot(4, 4, plot_idx)\n        if lead_name == \"II_long\": plt.subplot(4, 1, 4)\n            \n        plt.plot(signal_centered, color='blue', linewidth=1)\n        plt.title(f\"Dynamic {lead_name}\")\n        plt.grid(True, alpha=0.3)\n        plt.axis('off')\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n    \n    return signals_1d\n\n# --- RUN IT ---\nif 'final_leads' in locals():\n    print(\"Running Dynamic Tracker...\")\n    ecg_data = extract_1d_signals_dynamic(final_leads)\n    \n    # Overlay to check if it catches the high peaks now\n    overlay_extraction_on_images(final_leads, ecg_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T07:55:34.910607Z","iopub.execute_input":"2025-12-27T07:55:34.911536Z","iopub.status.idle":"2025-12-27T07:55:37.864873Z","shell.execute_reply.started":"2025-12-27T07:55:34.911505Z","shell.execute_reply":"2025-12-27T07:55:37.863674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nfrom scipy import signal as scipy_signal\n\ndef compare_extraction_vs_csv(leads_dict, extracted_dict, csv_path):\n    \"\"\"\n    Overlays the Extracted Signal (Green) and Ground Truth CSV (Red) \n    on the original image to visualize the difference.\n    \"\"\"\n    # 1. Load Ground Truth\n    try:\n        df = pd.read_csv(csv_path)\n    except Exception as e:\n        print(f\"Error loading CSV: {e}\")\n        return\n\n    # Map CSV column names to our dictionary keys\n    # CSV cols: I, II, III, aVR, aVL, aVF, V1, V2, V3, V4, V5, V6\n    # Note: The CSV likely has a full 'II' column (10000 samples). \n    # We need to slice it for the grid view vs the rhythm strip.\n    \n    lead_names = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    \n    plt.figure(figsize=(24, 16))\n    plot_idx = 1\n    \n    for name in lead_names:\n        if name not in leads_dict or name not in extracted_dict: continue\n        \n        # --- PREPARE DATA ---\n        # A. Get Extracted Signal (Pixels)\n        extracted_sig = extracted_dict[name]\n        img_w = len(extracted_sig)\n        \n        # B. Get Ground Truth (mV)\n        # Drop NaNs (some columns in CSV are padded with NaNs)\n        csv_sig = df[name].dropna().values\n        \n        # In the standard layout, the CSV column 'II' might be the full rhythm strip (10k samples).\n        # But for the grid image 'II', we only want the first segment (same length as 'I').\n        # Assuming all grid leads are roughly same duration (2.5s).\n        if name == 'II' and len(csv_sig) > 3000:\n             # Take the first segment (matching length of Lead I)\n             ref_len = len(df['I'].dropna().values)\n             csv_sig = csv_sig[:ref_len]\n\n        # --- ALIGNMENT & SCALING ---\n        \n        # 1. Resample CSV to match Image Width (Time Alignment)\n        # We stretch/shrink the CSV array to be exactly 'img_w' pixels long\n        csv_resampled = scipy_signal.resample(csv_sig, img_w)\n        \n        # 2. Normalize Both Signals (Voltage Alignment)\n        # Since we don't know the exact pixels/mV yet, we visually align them \n        # by matching their Min/Max ranges to the image height.\n        \n        # Get Image stats\n        img_h = leads_dict[name].shape[0]\n        center_y = img_h / 2\n        \n        # Scale Extracted (It's already in pixels, centered around 0)\n        # We just need to add center_y to plot it\n        plot_extracted = -extracted_sig + center_y # Invert y-axis for plotting\n        \n        # Scale CSV (mV -> Pixels)\n        # We calculate a scaling factor based on the data range\n        # Heuristic: Map the CSV range (e.g. -1mV to 1mV) to fill 60% of image height\n        csv_range = np.max(csv_resampled) - np.min(csv_resampled)\n        if csv_range == 0: csv_range = 1\n        \n        scale_factor = (img_h * 0.6) / csv_range\n        \n        # Center CSV around 0, scale it, then move to image center\n        csv_centered = csv_resampled - np.mean(csv_resampled)\n        plot_csv = -(csv_centered * scale_factor) + center_y\n\n        # --- VISUALIZATION ---\n        \n        # Convert grayscale image to color so we can draw colored lines\n        original_img = leads_dict[name]\n        if len(original_img.shape) == 2:\n            color_img = cv2.cvtColor(original_img, cv2.COLOR_GRAY2BGR)\n        else:\n            color_img = original_img.copy()\n\n        # Draw Lines (using Matplotlib is easier than CV2 for overlay consistency)\n        plt.subplot(3, 4, plot_idx)\n        plt.imshow(color_img)\n        \n        # Plot Extracted (Green)\n        plt.plot(plot_extracted, color='#00FF00', linewidth=2, label='Extracted', alpha=0.8)\n        \n        # Plot Ground Truth (Red)\n        plt.plot(plot_csv, color='red', linewidth=2, label='Ground Truth (CSV)', alpha=0.8, linestyle='--')\n        \n        plt.title(f\"{name} Comparison\")\n        plt.axis('off')\n        if plot_idx == 1: plt.legend(loc='upper left') # Show legend only on first\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n\n# --- RUN IT ---\n# Upload your CSV file to the environment first!\ncsv_filename = \"/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939.csv\" # Ensure this matches your uploaded file name\n\nif 'final_leads' in locals() and 'ecg_data' in locals():\n    compare_extraction_vs_csv(final_leads, ecg_data, csv_filename)\nelse:\n    print(\"Please run the Extraction step first to generate 'ecg_data'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T08:02:28.099918Z","iopub.execute_input":"2025-12-27T08:02:28.100247Z","iopub.status.idle":"2025-12-27T08:02:29.883627Z","shell.execute_reply.started":"2025-12-27T08:02:28.100202Z","shell.execute_reply":"2025-12-27T08:02:29.882304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport cv2\nfrom scipy import signal as scipy_signal\n\n# --- 1. THE EXTRACTION FUNCTION (EARLIER VERSION: BLOB FILTERING) ---\ndef extract_1d_signals_clean(leads_dict):\n    \"\"\"\n    Digitizes ECG signals by identifying large connected components (the wave)\n    and removing small components (text/noise). Robust to high voltages.\n    \"\"\"\n    signals_1d = {}\n    \n    # Standard plotting order\n    standard_order = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    if \"II_long\" in leads_dict: standard_order.append(\"II_long\")\n\n    print(\"Extracting signals using Connected Components (Text Removal)...\")\n    \n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n            \n        img = leads_dict[lead_name]\n        \n        # 1. Preprocessing\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else:\n            gray = img\n            \n        # Threshold: Signal (Black) -> 255 (White)\n        _, binary = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n        \n        # 2. Connected Components (The \"Blob\" Logic)\n        num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(binary, connectivity=8)\n        \n        # Create a mask for ONLY the signal\n        clean_mask = np.zeros_like(binary)\n        \n        for i in range(1, num_labels):\n            width = stats[i, cv2.CC_STAT_WIDTH]\n            area = stats[i, cv2.CC_STAT_AREA]\n            \n            # Keep only \"wide\" objects (signals) or very dense ones\n            # Text letters are usually narrow. Signals span the grid.\n            if width > (binary.shape[1] * 0.15): \n                clean_mask[labels == i] = 255\n                \n        # 3. Column-wise Extraction\n        h, w = clean_mask.shape\n        signal = []\n        \n        for col in range(w):\n            column_pixels = clean_mask[:, col]\n            signal_indices = np.where(column_pixels > 0)[0]\n            \n            if len(signal_indices) > 0:\n                y_center = np.median(signal_indices)\n                last_valid = h - y_center\n            else:\n                # If gap, hold previous value (linear interpolation would be better, but hold is safe)\n                last_valid = signal[-1] if len(signal) > 0 else h // 2\n                \n            signal.append(last_valid)\n            \n        # 4. Zero-Centering\n        signal = np.array(signal)\n        signal_centered = signal - np.median(signal)\n        signals_1d[lead_name] = signal_centered\n        \n    return signals_1d\n\n# --- 2. THE COMPARISON FUNCTION (CYAN VS RED) ---\ndef compare_with_csv_overlay(leads_dict, extracted_dict, csv_path):\n    try:\n        df = pd.read_csv(csv_path)\n    except Exception as e:\n        print(f\"Error loading CSV: {e}\")\n        return\n\n    lead_names = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    \n    plt.figure(figsize=(24, 16))\n    plot_idx = 1\n    \n    for name in lead_names:\n        if name not in leads_dict or name not in extracted_dict: continue\n        \n        # Data Setup\n        extracted_sig = extracted_dict[name]\n        img_w = len(extracted_sig)\n        csv_sig = df[name].dropna().values\n        \n        # Handle Rhythm Strip length mismatch for 'II'\n        if name == 'II' and len(csv_sig) > 3000:\n             ref_len = len(df['I'].dropna().values) # Use length of Lead I as reference\n             csv_sig = csv_sig[:ref_len]\n\n        # 1. Resample CSV to match Image Width (Time Alignment)\n        csv_resampled = scipy_signal.resample(csv_sig, img_w)\n        \n        # 2. Scaling (Voltage Alignment)\n        img_h = leads_dict[name].shape[0]\n        center_y = img_h / 2\n        \n        # Extracted (Cyan) - already in pixels\n        plot_extracted = -extracted_sig + center_y \n        \n        # CSV (Red) - mV to Pixels\n        csv_range = np.max(csv_resampled) - np.min(csv_resampled)\n        if csv_range == 0: csv_range = 1\n        scale_factor = (img_h * 0.7) / csv_range # Fill 70% of box\n        \n        csv_centered = csv_resampled - np.mean(csv_resampled)\n        plot_csv = -(csv_centered * scale_factor) + center_y\n\n        # 3. Plotting\n        original_img = leads_dict[name]\n        if len(original_img.shape) == 2:\n            color_img = cv2.cvtColor(original_img, cv2.COLOR_GRAY2BGR)\n        else:\n            color_img = original_img.copy()\n\n        plt.subplot(3, 4, plot_idx)\n        plt.imshow(color_img)\n        \n        # CYAN LINE = Your Extraction\n        plt.plot(plot_extracted, color='cyan', linewidth=2.5, label='Extracted (Image)', alpha=0.9)\n        \n        # RED LINE = Ground Truth CSV\n        plt.plot(plot_csv, color='red', linewidth=2, label='Truth (CSV)', alpha=0.7, linestyle='--')\n        \n        plt.title(f\"{name}\")\n        plt.axis('off')\n        if plot_idx == 1: plt.legend(loc='upper left', fontsize='small')\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n\n# --- EXECUTION ---\n# 1. Define CSV path\ncsv_filename = \"/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939.csv\" # <--- MAKE SURE THIS MATCHES THE IMAGE YOU ARE TESTING\n\n# 2. Run Extraction (Earlier Version)\nif 'final_leads' in locals():\n    ecg_data_clean = extract_1d_signals_clean(final_leads)\n\n    # 3. Compare\n    print(f\"Comparing Extraction vs {csv_filename}...\")\n    compare_with_csv_overlay(final_leads, ecg_data_clean, csv_filename)\nelse:\n    print(\"Error: 'final_leads' not found. Please run the segmentation steps first.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T08:05:21.303915Z","iopub.execute_input":"2025-12-27T08:05:21.304210Z","iopub.status.idle":"2025-12-27T08:05:23.441195Z","shell.execute_reply.started":"2025-12-27T08:05:21.304190Z","shell.execute_reply":"2025-12-27T08:05:23.439624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport pandas as pd\nfrom scipy import signal as scipy_signal\n\n# --- 1. THE MAGNETIC SNAKE TRACKER ALGORITHM ---\ndef extract_1d_signals_tracker(leads_dict):\n    \"\"\"\n    Digitizes ECG signals using a 'Snake Tracker' algorithm that follows \n    the line continuity to ignore text and isolated noise.\n    \"\"\"\n    signals_1d = {}\n    \n    # Standard plotting order\n    standard_order = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    if \"II_long\" in leads_dict: standard_order.append(\"II_long\")\n\n    print(\"Running Magnetic Snake Tracker (Stateful Tracking)...\")\n    \n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n            \n        img = leads_dict[lead_name]\n        \n        # 1. Preprocessing (Aggressive Grid Removal)\n        if len(img.shape) == 3:\n            gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else:\n            gray = img\n            \n        # Threshold: Invert so Signal = 255 (White), Background = 0 (Black)\n        _, binary = cv2.threshold(gray, 100, 255, cv2.THRESH_BINARY_INV)\n        \n        # 2. Morphological Cleaning (Connect the Snake)\n        kernel_h = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 1))\n        binary = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, kernel_h)\n        \n        h, w = binary.shape\n        signal = np.zeros(w)\n        \n        # --- THE SNAKE TRACKER ---\n        \n        # A. Find a starting point (Scan the first 10% of columns)\n        start_candidates = []\n        for col in range(int(w * 0.1)):\n            indices = np.where(binary[:, col] > 0)[0]\n            if len(indices) > 0:\n                start_candidates.extend(indices)\n        \n        if start_candidates:\n            current_y = np.median(start_candidates)\n        else:\n            current_y = h // 2 # Fallback\n            \n        # B. Walk through the image\n        search_radius = 15 # Look +/- 15 pixels up/down\n        \n        for col in range(w):\n            # Get all signal pixels in this column\n            col_pixels = np.where(binary[:, col] > 0)[0]\n            \n            if len(col_pixels) > 0:\n                # Filter pixels: Keep only those close to our current path\n                candidates = col_pixels[np.abs(col_pixels - current_y) < search_radius]\n                \n                if len(candidates) > 0:\n                    # Move to their center\n                    new_y = np.median(candidates)\n                    current_y = new_y # Update state\n                else:\n                    # No pixels nearby? HOLD position (don't jump to noise)\n                    pass \n            \n            # Record the value (Inverted for Graph: Height - Y)\n            signal[col] = h - current_y\n\n        # 3. Post-Processing\n        signal_centered = signal - np.median(signal)\n        signals_1d[lead_name] = signal_centered\n        \n    return signals_1d\n\n# --- 2. THE COMPARISON FUNCTION (CYAN VS RED) ---\ndef compare_with_csv_overlay(leads_dict, extracted_dict, csv_path):\n    try:\n        df = pd.read_csv(csv_path)\n    except Exception as e:\n        print(f\"Error loading CSV: {e}\")\n        return\n\n    lead_names = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    \n    plt.figure(figsize=(24, 16))\n    plot_idx = 1\n    \n    for name in lead_names:\n        if name not in leads_dict or name not in extracted_dict: continue\n        \n        # Data Setup\n        extracted_sig = extracted_dict[name]\n        img_w = len(extracted_sig)\n        csv_sig = df[name].dropna().values\n        \n        # Handle Rhythm Strip length mismatch for 'II'\n        if name == 'II' and len(csv_sig) > 3000:\n             ref_len = len(df['I'].dropna().values)\n             csv_sig = csv_sig[:ref_len]\n\n        # 1. Resample CSV to match Image Width (Time Alignment)\n        csv_resampled = scipy_signal.resample(csv_sig, img_w)\n        \n        # 2. Scaling (Voltage Alignment)\n        img_h = leads_dict[name].shape[0]\n        center_y = img_h / 2\n        \n        plot_extracted = -extracted_sig + center_y \n        \n        csv_range = np.max(csv_resampled) - np.min(csv_resampled)\n        if csv_range == 0: csv_range = 1\n        scale_factor = (img_h * 0.7) / csv_range\n        \n        csv_centered = csv_resampled - np.mean(csv_resampled)\n        plot_csv = -(csv_centered * scale_factor) + center_y\n\n        # 3. Plotting\n        original_img = leads_dict[name]\n        if len(original_img.shape) == 2:\n            color_img = cv2.cvtColor(original_img, cv2.COLOR_GRAY2BGR)\n        else:\n            color_img = original_img.copy()\n\n        plt.subplot(3, 4, plot_idx)\n        plt.imshow(color_img)\n        \n        # CYAN LINE = Your Tracker Extraction\n        plt.plot(plot_extracted, color='cyan', linewidth=2.5, label='Tracker (Image)', alpha=0.9)\n        \n        # RED LINE = Ground Truth CSV\n        plt.plot(plot_csv, color='red', linewidth=2, label='Truth (CSV)', alpha=0.7, linestyle='--')\n        \n        plt.title(f\"{name}\")\n        plt.axis('off')\n        if plot_idx == 1: plt.legend(loc='upper left', fontsize='small')\n        \n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n\n# --- EXECUTION ---\n# 1. Define CSV path\ncsv_filename = \"/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939.csv\" \n\n# 2. Run Snake Tracker\nif 'final_leads' in locals():\n    # Use the Tracker function\n    ecg_data_tracker = extract_1d_signals_tracker(final_leads)\n\n    # 3. Compare\n    print(f\"Comparing Snake Tracker vs {csv_filename}...\")\n    compare_with_csv_overlay(final_leads, ecg_data_tracker, csv_filename)\nelse:\n    print(\"Error: 'final_leads' not found. Please run the segmentation steps first.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T08:06:09.609066Z","iopub.execute_input":"2025-12-27T08:06:09.609422Z","iopub.status.idle":"2025-12-27T08:06:11.880561Z","shell.execute_reply.started":"2025-12-27T08:06:09.609396Z","shell.execute_reply":"2025-12-27T08:06:11.879494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport pandas as pd\nfrom scipy import signal as scipy_signal\n\n# --- 1. SEGMENTATION FUNCTIONS (Required to get the leads first) ---\ndef slice_ecg_rows(image):\n    h, w = image.shape[:2]\n    # Standard cuts at 0, 25%, 50%, 75%, 100% height\n    cuts = [0, int(h*0.25), int(h*0.50), int(h*0.75), h]\n    rows = []\n    for i in range(4):\n        rows.append(image[cuts[i]:cuts[i+1], :])\n    return rows\n\ndef segment_leads_with_shift(row_images, shift_col_4=0):\n    lead_names_grid = [\n        [\"I\", \"aVR\", \"V1\", \"V4\"],\n        [\"II\", \"aVL\", \"V2\", \"V5\"],\n        [\"III\", \"aVF\", \"V3\", \"V6\"]\n    ]\n    final_leads = {}\n    for row_idx in range(min(3, len(row_images))):\n        current_row = row_images[row_idx]\n        h, w = current_row.shape[:2]\n        # Standard cuts at 0, 25%, 50%, 75%, 100% width\n        cuts = [0, int(w*0.25), int(w*0.50), int(w*0.75), w]\n        cuts[3] += shift_col_4 # Apply the manual shift for V4/V5/V6\n        \n        for col_idx in range(4):\n            x1, x2 = cuts[col_idx], cuts[col_idx+1]\n            final_leads[lead_names_grid[row_idx][col_idx]] = current_row[:, x1:x2]\n            \n    if len(row_images) > 3:\n        final_leads['II_long'] = row_images[3]\n    return final_leads\n\n# --- 2. YOUR EXTRACTION CODE (Code 1: Connected Components) ---\ndef extract_1d_signals_clean(leads_dict):\n    \"\"\"\n    Converts lead images into 1D signal arrays, REMOVING TEXT NOISE using Connected Components.\n    \"\"\"\n    signals_1d = {}\n    standard_order = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    if \"II_long\" in leads_dict: standard_order.append(\"II_long\")\n\n    print(\"Extracting signals using Connected Components...\")\n    \n    for lead_name in standard_order:\n        if lead_name not in leads_dict: continue\n        img = leads_dict[lead_name]\n        \n        # Preprocessing\n        if len(img.shape) == 3: gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        else: gray = img\n        _, binary = cv2.threshold(gray, 120, 255, cv2.THRESH_BINARY_INV)\n        \n        # Connected Components\n        num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(binary, connectivity=8)\n        clean_mask = np.zeros_like(binary)\n        \n        for i in range(1, num_labels):\n            width = stats[i, cv2.CC_STAT_WIDTH]\n            # Keep only \"wide\" objects (signals) to filter out text\n            if width > (binary.shape[1] * 0.15): \n                clean_mask[labels == i] = 255\n                \n        # Column-wise Extraction\n        h, w = clean_mask.shape\n        signal = []\n        for col in range(w):\n            column_pixels = clean_mask[:, col]\n            signal_indices = np.where(column_pixels > 0)[0]\n            \n            if len(signal_indices) > 0:\n                y_center = np.median(signal_indices)\n                last_valid = h - y_center\n            else:\n                last_valid = signal[-1] if len(signal) > 0 else h // 2\n            signal.append(last_valid)\n            \n        signal = np.array(signal)\n        signal_centered = signal - np.median(signal)\n        signals_1d[lead_name] = signal_centered\n        \n    return signals_1d\n\n# --- 3. YOUR COMPARISON CODE (Overlay) ---\ndef compare_extraction_vs_csv(leads_dict, extracted_dict, csv_path):\n    try:\n        df = pd.read_csv(csv_path)\n    except Exception as e:\n        print(f\"Error loading CSV: {e}\"); return\n\n    lead_names = [\"I\", \"II\", \"III\", \"aVR\", \"aVL\", \"aVF\", \"V1\", \"V2\", \"V3\", \"V4\", \"V5\", \"V6\"]\n    \n    plt.figure(figsize=(24, 16))\n    plot_idx = 1\n    \n    for name in lead_names:\n        if name not in leads_dict or name not in extracted_dict: continue\n        \n        # Data Prep\n        extracted_sig = extracted_dict[name]\n        img_w = len(extracted_sig)\n        csv_sig = df[name].dropna().values\n        \n        # Handle Rhythm Strip length mismatch\n        if name == 'II' and len(csv_sig) > 3000:\n             ref_len = len(df['I'].dropna().values)\n             csv_sig = csv_sig[:ref_len]\n\n        # Resample & Scale\n        csv_resampled = scipy_signal.resample(csv_sig, img_w)\n        img_h = leads_dict[name].shape[0]\n        center_y = img_h / 2\n        \n        plot_extracted = -extracted_sig + center_y \n        \n        csv_range = np.max(csv_resampled) - np.min(csv_resampled)\n        if csv_range == 0: csv_range = 1\n        scale_factor = (img_h * 0.6) / csv_range\n        csv_centered = csv_resampled - np.mean(csv_resampled)\n        plot_csv = -(csv_centered * scale_factor) + center_y\n\n        # Visualization\n        original_img = leads_dict[name]\n        if len(original_img.shape) == 2: color_img = cv2.cvtColor(original_img, cv2.COLOR_GRAY2BGR)\n        else: color_img = original_img.copy()\n\n        plt.subplot(3, 4, plot_idx)\n        plt.imshow(color_img)\n        \n        # Extracted (Green)\n        plt.plot(plot_extracted, color='#00FF00', linewidth=2, label='Extracted', alpha=0.8)\n        # Ground Truth (Red)\n        plt.plot(plot_csv, color='red', linewidth=2, label='Ground Truth (CSV)', alpha=0.8, linestyle='--')\n        \n        plt.title(f\"{name} Comparison\")\n        plt.axis('off')\n        if plot_idx == 1: plt.legend(loc='upper left')\n        plot_idx += 1\n\n    plt.tight_layout()\n    plt.show()\n\n# --- MAIN EXECUTION ---\n# 1. Load your specific image and CSV\nimg_path = '/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939-0001.png' \ncsv_filename = '/kaggle/input/physionet-ecg-image-digitization/train/1015663939/1015663939.csv' \n\nimg = cv2.imread(img_path)\nif img is not None:\n    # A. Segmentation\n    rows = slice_ecg_rows(img)\n    final_leads = segment_leads_with_shift(rows, shift_col_4=0)\n    \n    # B. Extraction (Code 1)\n    ecg_data = extract_1d_signals_clean(final_leads)\n    \n    # C. Comparison (Your Overlay Function)\n    compare_extraction_vs_csv(final_leads, ecg_data, csv_filename)\nelse:\n    print(f\"Image not found at {img_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T08:11:11.230093Z","iopub.execute_input":"2025-12-27T08:11:11.230449Z","iopub.status.idle":"2025-12-27T08:11:13.820242Z","shell.execute_reply.started":"2025-12-27T08:11:11.230422Z","shell.execute_reply":"2025-12-27T08:11:13.818790Z"}},"outputs":[],"execution_count":null}]}