{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Base paths (adjust if needed)\nBASE_PATH = Path(\"/kaggle/input/physionet-ecg-image-digitization\")\nTRAIN_DIR = BASE_PATH / \"train\"\nTEST_DIR  = BASE_PATH / \"test\"\n\ndef find_sample_paths(base_dir: Path, sample_id: str):\n    \"\"\"Return list of matching image paths for a sample id under base_dir.\"\"\"\n    sample_folder = base_dir / sample_id\n    if not sample_folder.exists():\n        return []\n    return sorted(sample_folder.glob(f\"{sample_id}-*.png\"))\n\ndef load_ecg_image(sample_id: str,\n                   index: int = 0,\n                   split: str = \"train\",\n                   to_gray: bool = False,\n                   resize_to: tuple | None = None,\n                   normalize: bool = False,\n                   threshold: bool = False):\n    \"\"\"\n    Load an ECG PNG from the dataset.\n    - sample_id: e.g. \"1006427285\"\n    - index: which image for that sample (0-based)\n    - split: \"train\" or \"test\"\n    - to_gray: convert to grayscale\n    - resize_to: (w, h) tuple to resize image (use None to keep original)\n    - normalize: scale pixels to [0,1] float\n    - threshold: apply Otsu thresholding (only if grayscale)\n    Returns (img, img_path: Path)\n    \"\"\"\n    base_dir = TRAIN_DIR if split.lower() == \"train\" else TEST_DIR\n    candidates = find_sample_paths(base_dir, sample_id)\n    if not candidates:\n        raise FileNotFoundError(f\"No images found for sample '{sample_id}' in {base_dir}\")\n    if index < 0 or index >= len(candidates):\n        raise IndexError(f\"Index {index} out of range for sample (found {len(candidates)} images).\")\n    img_path = candidates[index]\n    img = cv2.imread(str(img_path), cv2.IMREAD_UNCHANGED)\n    if img is None:\n        raise IOError(f\"cv2.imread failed to load {img_path}\")\n\n    # Optional conversions\n    if resize_to:\n        img = cv2.resize(img, tuple(resize_to), interpolation=cv2.INTER_AREA)\n\n    if to_gray:\n        # if image already single channel, keep as is\n        if img.ndim == 3:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    else:\n        # convert BGR -> RGB for display (if color)\n        if img.ndim == 3:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    if threshold:\n        if img.ndim != 2:\n            # threshold requires grayscale\n            img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY) if img.ndim == 3 else img\n        _, img = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    if normalize:\n        img = img.astype(\"float32\") / 255.0\n\n    return img, img_path\n\ndef show_image(img, title: str = None, cmap: str | None = None, figsize=(10,6)):\n    plt.figure(figsize=figsize)\n    if img.ndim == 2:\n        plt.imshow(img, cmap=cmap or \"gray\", aspect=\"auto\")\n    else:\n        plt.imshow(img, aspect=\"auto\")\n    if title:\n        plt.title(title)\n    plt.axis(\"off\")\n    plt.show()\n\n# ---------------------------\n# Example usage with your sample_id\n# ---------------------------\nsample_id = \"1006427285\"\n\ntry:\n    img, p = load_ecg_image(sample_id, index=0, split=\"train\", to_gray=False, resize_to=None)\n    print(\"Loaded:\", p)\n    print(\"Shape:\", img.shape, \"dtype:\", img.dtype, \"min/max:\", img.min(), img.max())\n    show_image(img, title=f\"{sample_id} (original)\")\nexcept Exception as e:\n    print(\"Error:\", e)\n    # Show available candidates if any\n    candidates = find_sample_paths(TRAIN_DIR, sample_id)\n    if candidates:\n        print(\"Available files:\")\n        for c in candidates:\n            print(\" -\", c)\n    else:\n        print(f\"No folder for {sample_id} under {TRAIN_DIR}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:11:46.293585Z","iopub.execute_input":"2025-11-15T22:11:46.293817Z","iopub.status.idle":"2025-11-15T22:11:47.557762Z","shell.execute_reply.started":"2025-11-15T22:11:46.293796Z","shell.execute_reply":"2025-11-15T22:11:47.556638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# --- Paths ---\nBASE_PATH = Path(\"/kaggle/input/physionet-ecg-image-digitization\")\nTRAIN_DIR = BASE_PATH / \"train\"\n\nsample_id = \"1006427285\"\nimg_name = f\"{sample_id}-0001.png\"\nimg_path = TRAIN_DIR / sample_id / img_name\n\nprint(\"Image path:\", img_path)\n\n# --- Load image safely ---\nif not img_path.exists():\n    raise FileNotFoundError(f\"Image not found: {img_path}\")\n\nimg_bgr = cv2.imread(str(img_path), cv2.IMREAD_COLOR)\nif img_bgr is None:\n    raise RuntimeError(\"cv2.imread failed — image is unreadable.\")\n\n# Convert BGR → RGB (for matplotlib)\nimg_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n\n# Convert to grayscale\ngray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\n\n# --- Debug info ---\nprint(\"RGB shape:\", img_rgb.shape, \"min/max:\", img_rgb.min(), img_rgb.max())\nprint(\"Gray shape:\", gray.shape, \"min/max:\", gray.min(), gray.max())\n\n# --- Display ---\nplt.figure(figsize=(12, 6))\nplt.imshow(gray, cmap='gray')\nplt.title(f\"ECG Image: {img_name}\", fontsize=14)\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:12:44.262484Z","iopub.execute_input":"2025-11-15T22:12:44.263426Z","iopub.status.idle":"2025-11-15T22:12:44.786393Z","shell.execute_reply.started":"2025-11-15T22:12:44.263397Z","shell.execute_reply":"2025-11-15T22:12:44.785446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# --- Parameters you can tune ---\nGAUSSIAN_KSIZE = (3, 3)\nCANNY_MIN, CANNY_MAX = 50, 150\nMORPH_KERNEL = (3, 3)       # used to strengthen thin grid lines\nHOUGH_RHO = 1\nHOUGH_THETA = np.pi / 180\nHOUGH_THRESHOLD = 100       # minimum votes (increase to detect only strongest lines)\nMIN_LINE_LENGTH = 200       # pixels (reduce if grid squares are small)\nMAX_LINE_GAP = 10           # allowed gap for HoughLinesP linking\nANGLE_TOL_DEG = 5           # degrees tolerance to accept near-horizontal/vertical\n\n# --- Load image if only path given (optional) ---\n# If you already have img_bgr from previous cell, this will be skipped.\ntry:\n    img_bgr\nexcept NameError:\n    img_path = Path(\"/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285-0001.png\")\n    if not img_path.exists():\n        raise FileNotFoundError(img_path)\n    img_bgr = cv2.imread(str(img_path), cv2.IMREAD_COLOR)\n\n# --- Preprocess ---\ngray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\nblur = cv2.GaussianBlur(gray, GAUSSIAN_KSIZE, 0)\n\n# strengthen linear structures: morphological closing then opening can help\nkernel = cv2.getStructuringElement(cv2.MORPH_RECT, MORPH_KERNEL)\nmorph = cv2.morphologyEx(blur, cv2.MORPH_CLOSE, kernel)\nmorph = cv2.morphologyEx(morph, cv2.MORPH_OPEN, kernel)\n\nedges = cv2.Canny(morph, CANNY_MIN, CANNY_MAX, apertureSize=3)\n\n# --- Hough (probabilistic) ---\nlines_p = cv2.HoughLinesP(edges,\n                          rho=HOUGH_RHO,\n                          theta=HOUGH_THETA,\n                          threshold=HOUGH_THRESHOLD,\n                          minLineLength=MIN_LINE_LENGTH,\n                          maxLineGap=MAX_LINE_GAP)\n\n# prepare overlay image\noverlay = img_bgr.copy()\nout_lines = []\n\nif lines_p is None:\n    print(\"No lines found by HoughLinesP. Try reducing HOUGH_THRESHOLD or MIN_LINE_LENGTH.\")\nelse:\n    # filter lines near horizontal or vertical\n    for x1, y1, x2, y2 in lines_p.reshape(-1, 4):\n        dx = x2 - x1\n        dy = y2 - y1\n        angle_deg = abs(np.degrees(np.arctan2(dy, dx)))\n        angle_norm = min(angle_deg, 180 - angle_deg)  # map to [0,90]\n        # accept if angle close to 0 (horizontal) or 90 (vertical)\n        if angle_norm <= ANGLE_TOL_DEG or (90 - angle_norm) <= ANGLE_TOL_DEG:\n            out_lines.append((x1, y1, x2, y2, angle_deg))\n\n    print(f\"Total Hough segments: {len(lines_p)}. Kept after angle filtering: {len(out_lines)}\")\n\n    # draw filtered lines on overlay\n    for (x1, y1, x2, y2, angle_deg) in out_lines:\n        # thicker line for visibility\n        cv2.line(overlay, (x1, y1), (x2, y2), (0, 255, 0), thickness=2, lineType=cv2.LINE_AA)\n\n# Blend overlay with original for translucency\nalpha = 0.7\ngrid_vis = cv2.addWeighted(overlay, alpha, img_bgr, 1 - alpha, 0)\n\n# Convert to RGB for matplotlib\ngrid_vis_rgb = cv2.cvtColor(grid_vis, cv2.COLOR_BGR2RGB)\nedges_rgb = cv2.cvtColor(cv2.cvtColor(edges, cv2.COLOR_GRAY2BGR), cv2.COLOR_BGR2RGB)\n\n# --- Display results ---\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\naxes[0].imshow(cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB))\naxes[0].set_title(\"Original (RGB)\")\naxes[0].axis(\"off\")\n\naxes[1].imshow(edges, cmap=\"gray\")\naxes[1].set_title(\"Canny Edges\")\naxes[1].axis(\"off\")\n\naxes[2].imshow(grid_vis_rgb)\naxes[2].set_title(\"Detected Grid Lines Overlay\")\naxes[2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:14:11.130743Z","iopub.execute_input":"2025-11-15T22:14:11.131097Z","iopub.status.idle":"2025-11-15T22:14:13.084025Z","shell.execute_reply.started":"2025-11-15T22:14:11.131071Z","shell.execute_reply":"2025-11-15T22:14:13.082993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom skimage.morphology import skeletonize\nfrom pathlib import Path\nfrom typing import Optional, Tuple\n\ndef preprocess_ecg_image(img,\n                         grid_kernel: Optional[int] = None,\n                         denoise_ksize: int = 3,\n                         close_iters: int = 2,\n                         open_iters: int = 1,\n                         show: bool = False,\n                         save_to: Optional[Path] = None) -> np.ndarray:\n    \"\"\"\n    Remove grid, denoise, morphologically enhance and skeletonize an ECG grayscale image.\n    Returns a uint8 image with values {0,255} (skeletonized waveform).\n    \n    Parameters\n    ----------\n    img : np.ndarray\n        Input image. Can be grayscale (H,W) or color (H,W,3).\n    grid_kernel : Optional[int]\n        Approximate grid spacing (px). If None, an adaptive estimate based on image size is used.\n    denoise_ksize : int\n        Gaussian blur kernel to reduce sensor/noise before processing.\n    close_iters, open_iters : int\n        Iteration counts for morphological closing/opening (on binary image).\n    show : bool\n        If True, plot intermediate steps.\n    save_to : Optional[pathlib.Path]\n        If provided, save the final skeleton image to this path.\n    \"\"\"\n    # --- Validate and prepare input ---\n    if img is None:\n        raise ValueError(\"Input image is None.\")\n    if img.ndim == 3:\n        img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    else:\n        img_gray = img.copy()\n\n    img_gray = img_gray.astype(np.uint8)\n\n    h, w = img_gray.shape\n    # Adaptive default for grid kernel: grid lines often ~ (h / 40) px for A4 scan; clamp to sensible range\n    if grid_kernel is None:\n        grid_kernel = max(9, min(60, int(round(h / 40))))\n    grid_kernel = int(grid_kernel) if grid_kernel > 1 else 3\n\n    # --- 0. gentle denoise to suppress tiny speckle noise ---\n    if denoise_ksize > 0:\n        blur = cv2.GaussianBlur(img_gray, (denoise_ksize, denoise_ksize), 0)\n    else:\n        blur = img_gray\n\n    # --- 1. Extract background grid (horizontal + vertical) using morphological open ---\n    horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (grid_kernel, 1))\n    vert_kernel  = cv2.getStructuringElement(cv2.MORPH_RECT, (1, grid_kernel))\n    horizontal = cv2.morphologyEx(blur, cv2.MORPH_OPEN, horiz_kernel)\n    vertical   = cv2.morphologyEx(blur, cv2.MORPH_OPEN, vert_kernel)\n\n    # Combine grid components & subtract from original (preserve intensity)\n    grid = cv2.add(horizontal, vertical)\n    no_grid = cv2.subtract(blur, grid)\n\n    # Optionally normalize contrast after grid removal to boost waveform\n    no_grid = cv2.normalize(no_grid, None, 0, 255, cv2.NORM_MINMAX)\n\n    # --- 2. Binarize (invert so waveform = white) and morphological filtering ---\n    _, binary = cv2.threshold(no_grid, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 3))\n    morph = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, kernel, iterations=close_iters)\n    morph = cv2.morphologyEx(morph, cv2.MORPH_OPEN, kernel, iterations=open_iters)\n\n    # Remove very small islands (noise)\n    contours, _ = cv2.findContours(morph.copy(), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    min_area = max(10, (h * w) // 100000)  # adapt min area\n    clean = np.zeros_like(morph)\n    for cnt in contours:\n        if cv2.contourArea(cnt) >= min_area:\n            cv2.drawContours(clean, [cnt], -1, 255, thickness=-1)\n\n    # --- 3. Skeletonize: skimage expects boolean image (True = foreground) ---\n    bw_bool = (clean > 0)\n    skeleton = skeletonize(bw_bool).astype(np.uint8) * 255\n\n    # --- Optional save ---\n    if save_to is not None:\n        save_to = Path(save_to)\n        save_to.parent.mkdir(parents=True, exist_ok=True)\n        cv2.imwrite(str(save_to), skeleton)\n\n    # --- Optional plotting of intermediate steps ---\n    if show:\n        fig, axs = plt.subplots(1, 5, figsize=(18, 4))\n        axs[0].imshow(img_gray, cmap='gray'); axs[0].set_title(\"Input Gray\"); axs[0].axis('off')\n        axs[1].imshow(blur, cmap='gray'); axs[1].set_title(\"Denoised\"); axs[1].axis('off')\n        axs[2].imshow(no_grid, cmap='gray'); axs[2].set_title(f\"No Grid (k={grid_kernel})\"); axs[2].axis('off')\n        axs[3].imshow(morph, cmap='gray'); axs[3].set_title(\"Morph (binary)\"); axs[3].axis('off')\n        axs[4].imshow(skeleton, cmap='gray'); axs[4].set_title(\"Skeleton\"); axs[4].axis('off')\n        plt.tight_layout(); plt.show()\n\n    return skeleton\n\n\n# ---------------------------\n# Example usage:\n# ---------------------------\n# img_bgr = cv2.imread('/kaggle/input/physionet-ecg-image-digitization/train/1006867983/1006867983-0001.png')\n# skeleton = preprocess_ecg_image(img_bgr, grid_kernel=None, denoise_ksize=3, show=True)\n# plt.figure(figsize=(10,4)); plt.imshow(skeleton, cmap='gray'); plt.axis('off'); plt.title(\"Result\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:16:47.683574Z","iopub.execute_input":"2025-11-15T22:16:47.683941Z","iopub.status.idle":"2025-11-15T22:16:48.036319Z","shell.execute_reply.started":"2025-11-15T22:16:47.683917Z","shell.execute_reply":"2025-11-15T22:16:48.035373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom skimage.morphology import skeletonize\nfrom pathlib import Path\nfrom typing import Optional, Tuple\n\ndef extract_ecg_trace(img_bgr: np.ndarray,\n                      thin: Optional[np.ndarray] = None,\n                      tophat_k: Optional[int] = 7,\n                      adapt_block: int = 55,\n                      adapt_C: int = -5,\n                      grid_len_frac: float = 0.06,\n                      min_contour_area: int = 10,\n                      do_skeletonize: bool = True,\n                      show: bool = True) -> np.ndarray:\n    \"\"\"\n    Extract ECG trace (binary image) from a BGR input using combined techniques:\n      - top-hat enhancement\n      - global Otsu + adaptive thresholding\n      - morphological grid suppression\n      - contour extraction and filtering\n      - optional skeletonize to 1-px width\n\n    Returns uint8 binary image (0/255) same shape as input gray.\n    \"\"\"\n    if img_bgr is None:\n        raise ValueError(\"img_bgr is None\")\n\n    # --- ensure grayscale input and uint8 dtype ---\n    if img_bgr.ndim == 3:\n        img_gray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\n    else:\n        img_gray = img_bgr.copy()\n    img_gray = (np.clip(img_gray, 0, 255)).astype(np.uint8)\n    h, w = img_gray.shape\n\n    # --- ensure reasonable default kernel sizes proportional to image size ---\n    if tophat_k is None:\n        tophat_k = max(3, int(round(min(h, w) * 0.01)))  # ~1% of min dimension\n    if tophat_k % 2 == 0:\n        tophat_k += 1\n\n    # 1) Top-hat: enhance bright thin structures (use small kernel)\n    tkernel = cv2.getStructuringElement(cv2.MORPH_RECT, (tophat_k, tophat_k))\n    tophat = cv2.morphologyEx(img_gray, cv2.MORPH_TOPHAT, tkernel)\n\n    # 2) Alternative enhancement for dark traces: invert original and combine\n    inv = 255 - img_gray\n    enhanced = cv2.addWeighted(tophat, 0.6, inv, 0.4, 0)\n\n    # 3) Otsu global threshold (on a blurred version for stability)\n    blur = cv2.GaussianBlur(enhanced, (5, 5), 0)\n    _, binary_otsu = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # 4) Adaptive threshold (non-inverted white=trace)\n    # ensure blockSize is odd and >=3\n    block = adapt_block if adapt_block % 2 == 1 else adapt_block + 1\n    block = max(3, block)\n    bin_adapt = cv2.adaptiveThreshold(tophat, 255,\n                                      cv2.ADAPTIVE_THRESH_MEAN_C,\n                                      cv2.THRESH_BINARY,\n                                      block, adapt_C)\n\n    # Combine Otsu and adaptive results to benefit from both (logical OR)\n    combined = cv2.bitwise_or(binary_otsu, bin_adapt)\n\n    # 5) Grid suppression: remove long horizontal / vertical structures\n    # kernel lengths proportional to image size\n    grid_len = max(15, int(round(max(h, w) * grid_len_frac)))\n    hker = cv2.getStructuringElement(cv2.MORPH_RECT, (grid_len, 1))\n    vker = cv2.getStructuringElement(cv2.MORPH_RECT, (1, grid_len))\n    grid_h = cv2.morphologyEx(combined, cv2.MORPH_OPEN, hker)\n    grid_v = cv2.morphologyEx(combined, cv2.MORPH_OPEN, vker)\n    grid = cv2.bitwise_or(grid_h, grid_v)\n    no_grid = cv2.bitwise_and(combined, cv2.bitwise_not(grid))\n\n    # 6) Morphological closing to connect waveform pieces + small opening to remove speckles\n    kern3 = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 3))\n    processed = cv2.morphologyEx(no_grid, cv2.MORPH_CLOSE, kern3, iterations=2)\n    processed = cv2.morphologyEx(processed, cv2.MORPH_OPEN, kern3, iterations=1)\n\n    # 7) Contour extraction + filtering (remove tiny islands and extremely large background)\n    contours, _ = cv2.findContours(processed.copy(), cv2.RETR_LIST, cv2.CHAIN_APPROX_NONE)\n    wave = np.zeros_like(processed)\n    min_area = max(min_contour_area, (h * w) // 200000)  # adaptive min area\n    kept = 0\n    for cnt in contours:\n        area = cv2.contourArea(cnt)\n        if area >= min_area:\n            # Optionally filter by bounding box height (exclude huge boxes)\n            x, y, cw, ch = cv2.boundingRect(cnt)\n            if ch < h * 0.9:  # ignore contours covering almost whole height\n                cv2.drawContours(wave, [cnt], -1, 255, thickness=1)\n                kept += 1\n\n    # 8) If we requested skeletonization, thin to 1-px wide curve\n    if do_skeletonize:\n        bw_bool = (wave > 0)\n        sk = skeletonize(bw_bool).astype(np.uint8) * 255\n        final = sk\n    else:\n        final = wave\n\n    # 9) If a 'thin' input was provided, prefer it as final overlay option:\n    #    (not mandatory — only used as fallback if final is empty)\n    if thin is not None and np.count_nonzero(final) < 10:\n        t = thin.copy().astype(np.uint8)\n        if t.max() > 1:\n            t = (t > 0).astype(np.uint8) * 255\n        final = t\n\n    # --- Optional diagnostics plotting ---\n    if show:\n        fig, axs = plt.subplots(1, 5, figsize=(20, 4))\n        axs[0].imshow(img_gray, cmap='gray'); axs[0].set_title(\"Gray Input\"); axs[0].axis('off')\n        axs[1].imshow(tophat, cmap='gray'); axs[1].set_title(f\"Top-hat (k={tophat_k})\"); axs[1].axis('off')\n        axs[2].imshow(binary_otsu, cmap='gray'); axs[2].set_title(\"Otsu\"); axs[2].axis('off')\n        axs[3].imshow(bin_adapt, cmap='gray'); axs[3].set_title(\"Adaptive\"); axs[3].axis('off')\n        axs[4].imshow(final, cmap='gray'); axs[4].set_title(f\"Final Trace (kept {kept})\"); axs[4].axis('off')\n        plt.tight_layout(); plt.show()\n\n        # optional zoom of the extracted contours overlaid on the original\n        overlay = cv2.cvtColor(img_gray, cv2.COLOR_GRAY2BGR)\n        overlay[final > 0] = (0, 255, 0)  # green trace\n        plt.figure(figsize=(10, 4)); plt.imshow(cv2.cvtColor(overlay, cv2.COLOR_BGR2RGB))\n        plt.title(\"Overlay: extracted trace on original\"); plt.axis('off'); plt.show()\n\n    return final\n\n\n# -------------------------\n# Example usage:\n# -------------------------\n# if you have `thin` from previous skeleton result, pass it; else pass thin=None\n# img_bgr = cv2.imread(\"/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285-0001.png\")\n# result = extract_ecg_trace(img_bgr, thin=None, show=True)\n# plt.figure(figsize=(12,4)); plt.imshow(result, cmap='gray'); plt.axis('off'); plt.title(\"Extracted ECG Trace\"); plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:18:13.354089Z","iopub.execute_input":"2025-11-15T22:18:13.354505Z","iopub.status.idle":"2025-11-15T22:18:13.374037Z","shell.execute_reply.started":"2025-11-15T22:18:13.354469Z","shell.execute_reply":"2025-11-15T22:18:13.373162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom typing import List, Tuple\n\ndef refine_and_draw_contours(binary_in: np.ndarray,\n                             close_ksize: Tuple[int,int] = (5,5),\n                             close_iters: int = 2,\n                             contour_mode: int = cv2.RETR_EXTERNAL,\n                             approx_eps: float = 0.0,\n                             min_area: int | None = None,\n                             draw_thickness: int = 1,\n                             show: bool = True) -> Tuple[np.ndarray, List[np.ndarray]]:\n    \"\"\"\n    Take a binary image (foreground=white or black) and return an improved contour image.\n    - binary_in: input binary image (any dtype); nonzero = foreground\n    - close_ksize: morphological closing kernel size\n    - close_iters: iterations for closing\n    - contour_mode: cv2.RETR_* mode (default RETR_EXTERNAL to avoid nested hierarchy)\n    - approx_eps: if >0, approximate contours with cv2.approxPolyDP (epsilon in px)\n    - min_area: minimum contour area to keep (auto if None based on image size)\n    - draw_thickness: thickness to draw contours (1 for thin)\n    - show: display diagnostic plots\n    Returns (wave_img, contours_kept)\n    \"\"\"\n    if binary_in is None:\n        raise ValueError(\"binary_in is None\")\n\n    # Normalize to uint8 binary (0 or 255). Treat any nonzero as foreground.\n    bin_u8 = (binary_in > 0).astype(np.uint8) * 255\n\n    # Morphological closing to connect broken strokes\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, close_ksize)\n    closed = cv2.morphologyEx(bin_u8, cv2.MORPH_CLOSE, kernel, iterations=close_iters)\n\n    # We want white-on-black for contour detection (cv2 expects white foreground)\n    # Ensure foreground is white: closed already has 255 for foreground.\n    bin_for_contour = closed.copy()\n\n    # Detect contours (choose mode: RETR_EXTERNAL or RETR_TREE depending on needs)\n    contours, hierarchy = cv2.findContours(bin_for_contour, contour_mode, cv2.CHAIN_APPROX_NONE)\n\n    # Adaptive min_area if not provided\n    h, w = bin_for_contour.shape\n    if min_area is None:\n        min_area = max(10, (h * w) // 200000)  # small images -> small threshold\n\n    # Filter contours: remove tiny specks and ones that are almost whole-image\n    kept_contours = []\n    for cnt in contours:\n        area = cv2.contourArea(cnt)\n        if area < min_area:\n            continue\n        x, y, cw, ch = cv2.boundingRect(cnt)\n        if cw >= 0.95 * w and ch >= 0.95 * h:\n            # likely a full-frame object (ignore)\n            continue\n        if approx_eps and approx_eps > 0:\n            peri = cv2.arcLength(cnt, True)\n            eps_px = min(approx_eps, 0.02 * peri)  # cap to 2% of perimeter\n            cnt = cv2.approxPolyDP(cnt, eps_px, True)\n        kept_contours.append(cnt)\n\n    # Draw kept contours on a fresh blank canvas\n    wave = np.zeros_like(bin_for_contour)\n    if kept_contours:\n        cv2.drawContours(wave, kept_contours, -1, 255, thickness=draw_thickness, lineType=cv2.LINE_AA)\n\n    # Diagnostics / display\n    if show:\n        fig, ax = plt.subplots(1, 3, figsize=(18, 5))\n        ax[0].imshow(binary_in, cmap='gray'); ax[0].set_title(\"Input Binary (raw)\"); ax[0].axis('off')\n        ax[1].imshow(bin_for_contour, cmap='gray'); ax[1].set_title(\"After Closing (for contours)\"); ax[1].axis('off')\n        ax[2].imshow(wave, cmap='gray'); ax[2].set_title(f\"Final Contours (kept={len(kept_contours)})\"); ax[2].axis('off')\n        plt.tight_layout(); plt.show()\n\n        # overlay on original grayscale if available (try to infer)\n        try:\n            # if binary_in was from an original gray of same shape, overlay\n            overlay = cv2.cvtColor((binary_in > 0).astype(np.uint8) * 200, cv2.COLOR_GRAY2BGR)\n            overlay[w > 0] = (0, 255, 0)  # green overlay where wave exists\n            plt.figure(figsize=(8,3)); plt.imshow(overlay); plt.title(\"Overlay (approx)\"); plt.axis('off'); plt.show()\n        except Exception:\n            pass\n\n    return wave, kept_contours\n\n\n# ------------------------\n# Example usage:\n# ------------------------\n# assume `binary` is the result after your thresholding (np.ndarray)\n# wave_img, kept = refine_and_draw_contours(binary,\n#                                          close_ksize=(5,5),\n#                                          close_iters=2,\n#                                          contour_mode=cv2.RETR_EXTERNAL,\n#                                          approx_eps=2.0,\n#                                          min_area=None,\n#                                          draw_thickness=1,\n#                                          show=True)\n# print(\"Detected contour count:\", len(kept))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:19:02.480722Z","iopub.execute_input":"2025-11-15T22:19:02.481057Z","iopub.status.idle":"2025-11-15T22:19:02.495092Z","shell.execute_reply.started":"2025-11-15T22:19:02.481034Z","shell.execute_reply":"2025-11-15T22:19:02.494096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef draw_wave_contours(binary_input,\n                        thickness=3,\n                        min_area_factor=200000,\n                        show=True):\n    \"\"\"\n    binary_input : any binary-like or grayscale image.\n    Returns: wave (uint8) with contours drawn.\n    \"\"\"\n\n    if binary_input is None:\n        raise ValueError(\"binary_input = None — pass a valid image like binary_inv or binary.\")\n\n    # --- 1) Convert to uint8 binary 0/255 ---\n    if binary_input.ndim == 3 and binary_input.shape[-1] == 3:\n        gray = cv2.cvtColor(binary_input, cv2.COLOR_BGR2GRAY)\n    else:\n        gray = binary_input.copy()\n\n    gray = np.clip(gray, 0, 255).astype(np.uint8)\n\n    # if already binary-ish, collapse to 0/255\n    bin_u8 = (gray > 0).astype(np.uint8) * 255\n\n    # --- 2) Closing to solidify contours ---\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5,5))\n    closed = cv2.morphologyEx(bin_u8, cv2.MORPH_CLOSE, kernel, iterations=2)\n\n    # --- 3) Find contours (white = foreground) ---\n    contours, _ = cv2.findContours(closed, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n\n    h, w = closed.shape\n    min_area = max(10, (h*w)//min_area_factor)\n\n    filtered = []\n    for cnt in contours:\n        if cv2.contourArea(cnt) >= min_area:\n            filtered.append(cnt)\n\n    print(f\"Contours: {len(contours)}  → kept: {len(filtered)} (min_area={min_area})\")\n\n    # --- 4) Draw thick contours ---\n    wave = np.zeros_like(closed)\n    if filtered:\n        cv2.drawContours(wave, filtered, -1, 255, thickness, lineType=cv2.LINE_AA)\n\n    # --- 5) Show ---\n    if show:\n        plt.figure(figsize=(12,6))\n        plt.imshow(wave, cmap='gray')\n        plt.title(\"Contours (thickened)\")\n        plt.axis('off')\n        plt.show()\n\n    return wave\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:22:22.455682Z","iopub.execute_input":"2025-11-15T22:22:22.456116Z","iopub.status.idle":"2025-11-15T22:22:22.469433Z","shell.execute_reply.started":"2025-11-15T22:22:22.456070Z","shell.execute_reply":"2025-11-15T22:22:22.468299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef extract_wave_from_red_channel(\n        img,\n        blur_ksize=(5,5),\n        morph_kernel=(3,3),\n        morph_iters=1,\n        show=True\n    ):\n    \"\"\"\n    Extract ECG waveform by removing the red grid using the RED channel.\n    Output: binary image (white = waveform).\n    \"\"\"\n    if img is None:\n        raise ValueError(\"Input `img` is None.\")\n\n    # --- 1) Split channels (BGR) ---\n    if img.ndim == 3 and img.shape[2] == 3:\n        b, g, r = cv2.split(img)\n    else:\n        # fallback if grayscale\n        r = img.copy()\n\n    # --- 2) Invert red channel to make black waveform bright ---\n    wave_enhanced = cv2.subtract(255, r.astype(np.uint8))\n\n    # --- 3) Smooth before thresholding ---\n    blur = cv2.GaussianBlur(wave_enhanced, blur_ksize, 0)\n\n    # --- 4) Global Otsu binary threshold ---\n    _, binary = cv2.threshold(blur, 0, 255,\n                              cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # --- 5) Morphology cleanup (connect line, remove small noise) ---\n    k = cv2.getStructuringElement(cv2.MORPH_RECT, morph_kernel)\n    binary = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, k, iterations=morph_iters)\n    binary = cv2.morphologyEx(binary, cv2.MORPH_OPEN, k, iterations=1)\n\n    binary = (binary > 0).astype(np.uint8) * 255  # ensure clean binary\n\n    # --- 6) Visualize ---\n    if show:\n        fig, ax = plt.subplots(1, 3, figsize=(18, 5))\n        ax[0].imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n        ax[0].set_title(\"Original Image\")\n        ax[0].axis('off')\n\n        ax[1].imshow(wave_enhanced, cmap='gray')\n        ax[1].set_title(\"Red Channel Inverted (Wave Enhanced)\")\n        ax[1].axis('off')\n\n        ax[2].imshow(binary, cmap='gray')\n        ax[2].set_title(\"Binary Waveform (Otsu + Cleanup)\")\n        ax[2].axis('off')\n\n        plt.tight_layout()\n        plt.show()\n\n        # Overlay wave on original\n        try:\n            overlay = cv2.cvtColor(img.copy(), cv2.COLOR_BGR2RGB)\n            overlay[binary > 0] = (0, 255, 0)\n            plt.figure(figsize=(10,4))\n            plt.imshow(overlay)\n            plt.title(\"Overlay: Extracted Waveform (green)\")\n            plt.axis('off')\n            plt.show()\n        except:\n            pass\n\n    return binary\n\n\n# -----------------------\n# Example:\n# -----------------------\n# img = cv2.imread(\"/kaggle/input/.../ECG.png\")\n# binary = extract_wave_from_red_channel(img, show=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:24:08.835024Z","iopub.execute_input":"2025-11-15T22:24:08.835384Z","iopub.status.idle":"2025-11-15T22:24:08.846049Z","shell.execute_reply.started":"2025-11-15T22:24:08.835363Z","shell.execute_reply":"2025-11-15T22:24:08.845032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom typing import Optional, Tuple\n\ndef build_binary_from_ecg(img_bgr: Optional[np.ndarray] = None,\n                          img_path: Optional[str] = None,\n                          blur_ksize: Tuple[int,int] = (5,5),\n                          adapt_block: int = 55,\n                          adapt_C: int = -5,\n                          grid_len_frac: float = 0.06,\n                          morph_kernel: Tuple[int,int] = (3,3),\n                          morph_iters: int = 1,\n                          show: bool = True) -> Tuple[np.ndarray, np.ndarray]:\n    \"\"\"\n    Build a robust binary image from an ECG PNG.\n\n    Returns (binary, binary_inv)\n      - binary: white=waveform, uint8 0/255\n      - binary_inv: inverted (black=waveform), uint8 0/255\n    Provide either img_bgr (np.ndarray) or img_path (str).\n    \"\"\"\n    # --- load image if path given ---\n    if img_bgr is None:\n        if img_path is None:\n            raise ValueError(\"Provide img_bgr or img_path\")\n        p = Path(img_path)\n        if not p.exists():\n            raise FileNotFoundError(f\"File not found: {img_path}\")\n        img_bgr = cv2.imread(str(p), cv2.IMREAD_COLOR)\n        if img_bgr is None:\n            raise RuntimeError(\"cv2.imread failed to load the image.\")\n\n    # --- ensure bgr ---\n    if img_bgr.ndim != 3 or img_bgr.shape[2] != 3:\n        raise ValueError(\"img_bgr must be a BGR color image (H,W,3)\")\n\n    h, w = img_bgr.shape[:2]\n\n    # --- 1) Extract red channel; invert so black waveform -> bright ---\n    b, g, r = cv2.split(img_bgr)\n    red_inv = cv2.subtract(255, r)  # waveform becomes brighter\n\n    # --- 2) Slight blur for stable thresholding ---\n    blur = cv2.GaussianBlur(red_inv, blur_ksize, 0)\n\n    # --- 3) Global Otsu threshold on blurred inverted-red ---\n    _, bin_otsu = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # --- 4) Adaptive threshold on the top-hat (local variations) ---\n    # top-hat helps emphasize thin bright structures:\n    th_k = max(3, int(round(min(h,w)*0.01)))\n    if th_k % 2 == 0:\n        th_k += 1\n    topk = cv2.getStructuringElement(cv2.MORPH_RECT, (th_k, th_k))\n    tophat = cv2.morphologyEx(red_inv, cv2.MORPH_TOPHAT, topk)\n\n    block = adapt_block if adapt_block % 2 == 1 else adapt_block + 1\n    block = max(3, block)\n    bin_adapt = cv2.adaptiveThreshold(tophat, 255,\n                                      cv2.ADAPTIVE_THRESH_MEAN_C,\n                                      cv2.THRESH_BINARY,\n                                      block, adapt_C)\n\n    # --- 5) Combine global + adaptive (OR) to benefit both ---\n    combined = cv2.bitwise_or(bin_otsu, bin_adapt)\n\n    # --- 6) Grid suppression: remove long horizontal/vertical lines ---\n    grid_len = max(15, int(round(max(h,w) * grid_len_frac)))\n    hker = cv2.getStructuringElement(cv2.MORPH_RECT, (grid_len, 1))\n    vker = cv2.getStructuringElement(cv2.MORPH_RECT, (1, grid_len))\n    grid_h = cv2.morphologyEx(combined, cv2.MORPH_OPEN, hker)\n    grid_v = cv2.morphologyEx(combined, cv2.MORPH_OPEN, vker)\n    grid = cv2.bitwise_or(grid_h, grid_v)\n    no_grid = cv2.bitwise_and(combined, cv2.bitwise_not(grid))\n\n    # --- 7) Morphological cleanup ---\n    k = cv2.getStructuringElement(cv2.MORPH_RECT, morph_kernel)\n    cleaned = cv2.morphologyEx(no_grid, cv2.MORPH_CLOSE, k, iterations=morph_iters)\n    cleaned = cv2.morphologyEx(cleaned, cv2.MORPH_OPEN, k, iterations=1)\n\n    # --- 8) Remove tiny islands: keep contours above minimal area ---\n    contours, _ = cv2.findContours(cleaned.copy(), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    min_area = max(10, (h * w) // 200000)\n    mask = np.zeros_like(cleaned)\n    for cnt in contours:\n        if cv2.contourArea(cnt) >= min_area:\n            x,y,cw,ch = cv2.boundingRect(cnt)\n            # skip giant full-frame shapes\n            if not (cw >= 0.95*w and ch >= 0.95*h):\n                cv2.drawContours(mask, [cnt], -1, 255, thickness=-1)\n\n    final = (mask > 0).astype(np.uint8) * 255\n\n    # make sure white = waveform. If final mean is high (i.e., background white), invert.\n    if final.mean() > 127:\n        final = cv2.bitwise_not(final)\n\n    binary = final\n    binary_inv = cv2.bitwise_not(binary)\n\n    # --- show diagnostics ---\n    if show:\n        fig, axs = plt.subplots(1, 4, figsize=(20,5))\n        axs[0].imshow(cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)); axs[0].set_title(\"Original (RGB)\"); axs[0].axis('off')\n        axs[1].imshow(red_inv, cmap='gray'); axs[1].set_title(\"Inverted Red (wave up)\"); axs[1].axis('off')\n        axs[2].imshow(combined, cmap='gray'); axs[2].set_title(\"Combined Otsu+Adaptive\"); axs[2].axis('off')\n        axs[3].imshow(binary, cmap='gray'); axs[3].set_title(\"Final Binary (white=wave)\"); axs[3].axis('off')\n        plt.tight_layout(); plt.show()\n\n        # overlay\n        overlay = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        overlay[binary > 0] = (0,255,0)\n        plt.figure(figsize=(12,4)); plt.imshow(overlay); plt.title(\"Overlay: binary on original\"); plt.axis('off'); plt.show()\n\n    # return both for convenience and to ensure names in notebook when you assign them\n    return binary, binary_inv\n\n# --------------------------\n# Example usage:\n# --------------------------\n# If you already have an image array `img_bgr`, call:\n# binary, binary_inv = build_binary_from_ecg(img_bgr=img_bgr, show=True)\n\n# Or, if you have a path:\n# binary, binary_inv = build_binary_from_ecg(img_path=\"/kaggle/input/.../1006427285-0001.png\", show=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:33:26.525004Z","iopub.execute_input":"2025-11-15T22:33:26.525727Z","iopub.status.idle":"2025-11-15T22:33:26.544717Z","shell.execute_reply.started":"2025-11-15T22:33:26.525697Z","shell.execute_reply":"2025-11-15T22:33:26.543482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimg_path = \"/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285-0001.png\"\nimg = cv2.imread(img_path)\n\n# --- create a simple binary so downstream code works ---\nb, g, r = cv2.split(img)\nred_inv = 255 - r\n\nblur = cv2.GaussianBlur(red_inv, (5,5), 0)\n_, binary = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\nbinary = (binary > 0).astype(np.uint8) * 255\nbinary_inv = cv2.bitwise_not(binary)\n\nprint(\"Created variables: binary, binary_inv\")\nplt.imshow(binary, cmap='gray')\nplt.title(\"Binary ready\")\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:13:38.135486Z","iopub.execute_input":"2025-11-16T10:13:38.135845Z","iopub.status.idle":"2025-11-16T10:13:39.014333Z","shell.execute_reply.started":"2025-11-16T10:13:38.135817Z","shell.execute_reply":"2025-11-16T10:13:39.013330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\n\ndef process_selected_images(selected_images,\n                            out_dir=None,\n                            grid_kernel=30,\n                            denoise_ksize=3,\n                            close_iters=2,\n                            open_iters=1,\n                            show=False,\n                            save_png=True,\n                            overwrite=False):\n    \"\"\"\n    Process a list of image file paths with preprocess_ecg_image() and optionally save outputs.\n\n    Args:\n      selected_images: iterable of file paths (str or Path)\n      out_dir: optional directory to save skeleton outputs (Path or str)\n      grid_kernel, denoise_ksize, close_iters, open_iters: forwarded to preprocess_ecg_image\n      show: if True, display each result (slow for many images)\n      save_png: if True, save skeleton as PNG to out_dir\n      overwrite: whether to overwrite existing files\n\n    Returns:\n      results: list of dicts with keys: filepath, success, error, out_path (if saved), skeleton_shape, nz_pixels\n    \"\"\"\n    selected_images = [Path(p) for p in selected_images]\n    if out_dir is not None:\n        out_dir = Path(out_dir)\n        out_dir.mkdir(parents=True, exist_ok=True)\n\n    results = []\n    for img_path in tqdm(selected_images, desc=\"Processing images\"):\n        meta = {\"filepath\": str(img_path), \"success\": False, \"error\": None, \"out_path\": None, \"skeleton_shape\": None, \"nz_pixels\": 0}\n        try:\n            if not img_path.exists():\n                raise FileNotFoundError(f\"File not found: {img_path}\")\n\n            # read image (support grayscale or color)\n            img = cv2.imread(str(img_path), cv2.IMREAD_UNCHANGED)\n            if img is None:\n                raise RuntimeError(\"cv2.imread returned None (file unreadable or unsupported)\")\n\n            # convert to gray if necessary\n            if img.ndim == 3:\n                img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            else:\n                img_gray = img.copy()\n\n            # call your preprocessing func (assumed to return skeleton uint8 0/255)\n            thin = preprocess_ecg_image(img_gray,\n                                       grid_kernel=grid_kernel,\n                                       denoise_ksize=denoise_ksize,\n                                       close_iters=close_iters,\n                                       open_iters=open_iters,\n                                       show=False)  # keep preprocessing quiet; we will show below if requested\n\n            # ensure dtype and values\n            thin_u8 = (thin > 0).astype(np.uint8) * 255\n            meta[\"skeleton_shape\"] = thin_u8.shape\n            meta[\"nz_pixels\"] = int(np.count_nonzero(thin_u8))\n\n            # save if requested\n            if save_png and out_dir is not None:\n                out_name = img_path.stem + \"_skeleton.png\"\n                out_path = out_dir / out_name\n                if out_path.exists() and not overwrite:\n                    # don't overwrite; append a counter\n                    i = 1\n                    while (out_dir / f\"{img_path.stem}_skeleton_{i}.png\").exists():\n                        i += 1\n                    out_path = out_dir / f\"{img_path.stem}_skeleton_{i}.png\"\n                cv2.imwrite(str(out_path), thin_u8)\n                meta[\"out_path\"] = str(out_path)\n\n            # optional display\n            if show:\n                fig, ax = plt.subplots(1,2, figsize=(12,4))\n                ax[0].imshow(img_gray, cmap='gray'); ax[0].set_title(img_path.name); ax[0].axis('off')\n                ax[1].imshow(thin_u8, cmap='gray'); ax[1].set_title(\"Skeleton\"); ax[1].axis('off')\n                plt.tight_layout(); plt.show()\n\n            meta[\"success\"] = True\n\n        except Exception as e:\n            meta[\"error\"] = str(e)\n\n        results.append(meta)\n\n    # optionally save CSV summary to out_dir\n    if out_dir is not None and results:\n        df = pd.DataFrame(results)\n        summary_path = out_dir / \"processing_summary.csv\"\n        df.to_csv(summary_path, index=False)\n        print(\"Saved summary:\", summary_path)\n\n    return results\n\n# -----------------------------\n# Example usage:\n# -----------------------------\n# selected_images = [\"/kaggle/input/.../1006427285-0001.png\", ...]\n# out = process_selected_images(selected_images, out_dir=\"/kaggle/working/skeletons\",\n#                               grid_kernel=30, show=False, save_png=True)\n# print(out[:3])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:16:14.240209Z","iopub.execute_input":"2025-11-16T10:16:14.240533Z","iopub.status.idle":"2025-11-16T10:16:14.760868Z","shell.execute_reply.started":"2025-11-16T10:16:14.240512Z","shell.execute_reply":"2025-11-16T10:16:14.760148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport time\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\nfrom typing import Iterable, Optional, List, Dict, Tuple, Any\n\ndef _suggest_grid_kernel(h: int, w: int) -> int:\n    \"\"\"Heuristic grid kernel: smaller for small images, larger for big scans.\"\"\"\n    # grid lines often ~ 25-30 px for typical ECG scans at 25mm/s; scale with image height\n    k = max(9, min(80, int(round(min(h, w) / 40))))\n    if k % 2 == 0:\n        k += 1\n    return k\n\ndef process_selected_images_refined(\n        selected_images: Iterable[str],\n        out_dir: Optional[str] = None,\n        grid_kernel: Optional[int] = None,         # if None, auto per-image\n        denoise_ksize: int = 3,\n        close_iters: int = 2,\n        open_iters: int = 1,\n        save_png: bool = True,\n        save_npy: bool = False,\n        overwrite: bool = False,\n        show: bool = False,\n        max_workers: int = 4,\n        progress_desc: str = \"Processing images\"\n    ) -> Tuple[pd.DataFrame, List[Dict[str,Any]]]:\n    \"\"\"\n    Process a list of ECG image paths using `preprocess_ecg_image()`.\n\n    Returns (summary_df, results_list) where results_list is a list of dicts per image.\n\n    - selected_images: iterable of file paths\n    - out_dir: optional directory where skeletons (.png/.npy) and CSV summary are saved\n    - grid_kernel: if provided, used for all images; otherwise auto-estimated per image\n    - max_workers: if >1, uses ThreadPoolExecutor to process multiple images concurrently\n    \"\"\"\n    selected_images = [Path(p) for p in selected_images]\n    if out_dir:\n        out_dir = Path(out_dir)\n        out_dir.mkdir(parents=True, exist_ok=True)\n\n    results: List[Dict[str,Any]] = []\n\n    def _process_one(img_path: Path) -> Dict[str, Any]:\n        meta = {\n            \"filepath\": str(img_path),\n            \"filename\": img_path.name,\n            \"success\": False,\n            \"error\": None,\n            \"skeleton_path\": None,\n            \"npy_path\": None,\n            \"height\": None,\n            \"width\": None,\n            \"nz_pixels\": 0,\n            \"elapsed_s\": None\n        }\n        start = time.time()\n        try:\n            if not img_path.exists():\n                raise FileNotFoundError(f\"File not found: {img_path}\")\n\n            img = cv2.imread(str(img_path), cv2.IMREAD_UNCHANGED)\n            if img is None:\n                raise RuntimeError(\"cv2.imread returned None (unsupported or corrupted file)\")\n\n            # convert to grayscale if needed\n            if img.ndim == 3 and img.shape[-1] == 3:\n                img_gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n            else:\n                img_gray = img.copy()\n\n            h, w = img_gray.shape[:2]\n            meta[\"height\"], meta[\"width\"] = h, w\n\n            # decide grid kernel\n            effective_grid_kernel = grid_kernel if grid_kernel is not None else _suggest_grid_kernel(h, w)\n\n            # call your preprocess_ecg_image -> should return uint8 skeleton (0/255)\n            thin = preprocess_ecg_image(img_gray,\n                                       grid_kernel=effective_grid_kernel,\n                                       denoise_ksize=denoise_ksize,\n                                       close_iters=close_iters,\n                                       open_iters=open_iters,\n                                       show=False)\n\n            thin_u8 = (thin > 0).astype(np.uint8) * 255\n            meta[\"nz_pixels\"] = int(np.count_nonzero(thin_u8))\n\n            # save outputs\n            if out_dir:\n                stem = img_path.stem\n                if save_png:\n                    p_png = out_dir / f\"{stem}_skeleton.png\"\n                    if p_png.exists() and not overwrite:\n                        # find non-colliding name\n                        i = 1\n                        while (out_dir / f\"{stem}_skeleton_{i}.png\").exists():\n                            i += 1\n                        p_png = out_dir / f\"{stem}_skeleton_{i}.png\"\n                    cv2.imwrite(str(p_png), thin_u8)\n                    meta[\"skeleton_path\"] = str(p_png)\n\n                if save_npy:\n                    p_npy = out_dir / f\"{stem}_skeleton.npy\"\n                    if p_npy.exists() and not overwrite:\n                        i = 1\n                        while (out_dir / f\"{stem}_skeleton_{i}.npy\").exists():\n                            i += 1\n                        p_npy = out_dir / f\"{stem}_skeleton_{i}.npy\"\n                    np.save(str(p_npy), thin_u8)\n                    meta[\"npy_path\"] = str(p_npy)\n\n            meta[\"success\"] = True\n        except Exception as e:\n            meta[\"error\"] = repr(e)\n        finally:\n            meta[\"elapsed_s\"] = round(time.time() - start, 3)\n            return meta\n\n    # choose executor only if >1 worker\n    if max_workers is None or max_workers <= 1:\n        # sequential\n        for img_path in tqdm(selected_images, desc=progress_desc):\n            results.append(_process_one(img_path))\n    else:\n        # threaded (good for I/O and light CPU) — use ThreadPoolExecutor\n        with ThreadPoolExecutor(max_workers=max_workers) as ex:\n            future_to_path = {ex.submit(_process_one, p): p for p in selected_images}\n            for f in tqdm(as_completed(future_to_path), total=len(future_to_path), desc=progress_desc):\n                meta = f.result()\n                results.append(meta)\n\n    # build DataFrame in original input order (stable)\n    df = pd.DataFrame(results)\n    if out_dir:\n        summary_csv = out_dir / \"processing_summary.csv\"\n        df.to_csv(summary_csv, index=False)\n        print(\"Saved summary:\", summary_csv)\n\n    # Optionally show a small table for quick glance\n    display_df = df[[\"filename\", \"height\", \"width\", \"nz_pixels\", \"success\", \"error\", \"skeleton_path\"]].copy()\n    display(display_df.head(20))\n\n    return df, results\n\n# -----------------------\n# Example usage:\n# -----------------------\n# selected_images = [\n#     \"/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285-0001.png\",\n#     \"/kaggle/input/physionet-ecg-image-digitization/train/1006867983/1006867983-0001.png\"\n# ]\n#\n# df, meta_list = process_selected_images_refined(\n#     selected_images,\n#     out_dir=\"/kaggle/working/skeletons\",\n#     grid_kernel=None,          # auto per-image\n#     save_png=True,\n#     save_npy=False,\n#     overwrite=False,\n#     show=False,\n#     max_workers=4\n# )\n#\n# # Quick inspect first result\n# if not df.empty and df.loc[0, \"skeleton_path\"]:\n#     import matplotlib.pyplot as plt\n#     sk = cv2.imread(df.loc[0, \"skeleton_path\"], cv2.IMREAD_GRAYSCALE)\n#     plt.figure(figsize=(12,4)); plt.imshow(sk, cmap='gray'); plt.axis('off'); plt.title(df.loc[0, \"filename\"]); plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:19:47.140268Z","iopub.execute_input":"2025-11-16T10:19:47.141277Z","iopub.status.idle":"2025-11-16T10:19:47.159808Z","shell.execute_reply.started":"2025-11-16T10:19:47.141245Z","shell.execute_reply":"2025-11-16T10:19:47.158951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Full one-cell pipeline (fixed): PNG -> binary -> skeleton -> (x,y) CSV\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom skimage.morphology import skeletonize\nfrom pathlib import Path\n\n# ---------------------------\n# User editable: set path here\n# ---------------------------\nimg_path = \"/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285-0001.png\"\nout_csv = \"ecg_waveform_extracted.csv\"\n\n# ---------------------------\n# Helper / pipeline functions\n# ---------------------------\ndef build_binary_from_red_channel(img_bgr, \n                                  blur_ksize=(5,5),\n                                  adapt_block=55, adapt_C=-5,\n                                  grid_len_frac=0.06,\n                                  morph_kernel=(3,3), morph_iters=1):\n    \"\"\"Return binary (white = waveform) and binary_inv (black = waveform).\"\"\"\n    if img_bgr is None:\n        raise ValueError(\"img_bgr is None\")\n    # ensure color\n    if img_bgr.ndim != 3 or img_bgr.shape[2] != 3:\n        raise ValueError(\"img_bgr must be BGR color image\")\n    h, w = img_bgr.shape[:2]\n\n    # red channel inverted to make black waveform bright\n    b, g, r = cv2.split(img_bgr)\n    red_inv = cv2.subtract(255, r)\n\n    # blur + Otsu\n    blur = cv2.GaussianBlur(red_inv, blur_ksize, 0)\n    _, bin_otsu = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n\n    # top-hat + adaptive threshold to capture thin lines\n    th_k = max(3, int(round(min(h,w)*0.01)))\n    if th_k % 2 == 0: th_k += 1\n    topk = cv2.getStructuringElement(cv2.MORPH_RECT, (th_k, th_k))\n    tophat = cv2.morphologyEx(red_inv, cv2.MORPH_TOPHAT, topk)\n\n    block = adapt_block if adapt_block % 2 == 1 else adapt_block + 1\n    block = max(3, block)\n    bin_adapt = cv2.adaptiveThreshold(tophat, 255, cv2.ADAPTIVE_THRESH_MEAN_C,\n                                      cv2.THRESH_BINARY, block, adapt_C)\n\n    # combined\n    combined = cv2.bitwise_or(bin_otsu, bin_adapt)\n\n    # grid suppression via long horizontal/vertical open\n    grid_len = max(15, int(round(max(h,w) * grid_len_frac)))\n    hker = cv2.getStructuringElement(cv2.MORPH_RECT, (grid_len, 1))\n    vker = cv2.getStructuringElement(cv2.MORPH_RECT, (1, grid_len))\n    grid_h = cv2.morphologyEx(combined, cv2.MORPH_OPEN, hker)\n    grid_v = cv2.morphologyEx(combined, cv2.MORPH_OPEN, vker)\n    grid = cv2.bitwise_or(grid_h, grid_v)\n    no_grid = cv2.bitwise_and(combined, cv2.bitwise_not(grid))\n\n    # morphological cleanup\n    k = cv2.getStructuringElement(cv2.MORPH_RECT, morph_kernel)\n    cleaned = cv2.morphologyEx(no_grid, cv2.MORPH_CLOSE, k, iterations=morph_iters)\n    cleaned = cv2.morphologyEx(cleaned, cv2.MORPH_OPEN, k, iterations=1)\n\n    # remove tiny islands, keep meaningful components\n    contours, _ = cv2.findContours(cleaned.copy(), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    min_area = max(10, (h * w) // 200000)\n    mask = np.zeros_like(cleaned)\n    for cnt in contours:\n        if cv2.contourArea(cnt) >= min_area:\n            x,y,cw,ch = cv2.boundingRect(cnt)\n            if not (cw >= 0.95*w and ch >= 0.95*h):\n                cv2.drawContours(mask, [cnt], -1, 255, thickness=-1)\n\n    final = (mask > 0).astype(np.uint8) * 255\n    # ensure white = waveform\n    if final.mean() > 127:\n        final = cv2.bitwise_not(final)\n    return final, cv2.bitwise_not(final)\n\ndef skeletonize_binary(binary):\n    \"\"\"Expect binary uint8 0/255, return skeleton uint8 0/255.\"\"\"\n    bw_bool = (binary > 0)\n    sk = skeletonize(bw_bool).astype(np.uint8) * 255\n    return sk\n\ndef extract_waveform_from_skeleton(skel, method=\"median\", interpolate=True, smooth=True, smooth_window=7, spike_threshold=20):\n    \"\"\"Return pandas Series y[x] extracted from skeleton skel (0/255).\"\"\"\n    if skel is None:\n        raise ValueError(\"skel is None\")\n    sk = (skel > 0).astype(np.uint8)\n    H, W = sk.shape\n    y_positions = []\n    for x in range(W):\n        ys = np.where(sk[:, x] > 0)[0]\n        if len(ys) == 0:\n            y_positions.append(np.nan)\n            continue\n        if method == \"median\":\n            y_positions.append(float(np.median(ys)))\n        elif method == \"mean\":\n            y_positions.append(float(np.mean(ys)))\n        elif method == \"middle\":\n            y_positions.append(float((ys.min() + ys.max())/2))\n        else:\n            raise ValueError(\"method must be 'median'|'mean'|'middle'\")\n\n    y = pd.Series(y_positions)\n\n    if interpolate:\n        # replaced deprecated fillna(method=...) with bfill()/ffill()\n        y = y.interpolate().bfill().ffill()\n\n    # remove abrupt single-column spikes\n    if spike_threshold is not None and spike_threshold > 0:\n        y_clean = y.copy()\n        for i in range(1, len(y_clean)):\n            if abs(y_clean.iat[i] - y_clean.iat[i-1]) > spike_threshold:\n                y_clean.iat[i] = y_clean.iat[i-1]\n        y = y_clean\n\n    if smooth:\n        win = smooth_window if (smooth_window % 2 == 1) else (smooth_window + 1)\n        y = y.rolling(win, center=True, min_periods=1).mean()\n\n    return y\n\n# ---------------------------\n# Run pipeline\n# ---------------------------\np = Path(img_path)\nif not p.exists():\n    raise FileNotFoundError(f\"Image not found: {img_path}\")\n\nimg_bgr = cv2.imread(str(p), cv2.IMREAD_COLOR)\nif img_bgr is None:\n    raise RuntimeError(\"cv2 failed to load image\")\n\n# 1) build binary\nbinary, binary_inv = build_binary_from_red_channel(img_bgr,\n                                                   blur_ksize=(5,5),\n                                                   adapt_block=55, adapt_C=-5,\n                                                   grid_len_frac=0.06,\n                                                   morph_kernel=(3,3), morph_iters=1)\n\n# 2) skeletonize -> this variable is guaranteed to exist\nskeleton = skeletonize_binary(binary)\n\n# 3) extract waveform series\ny_series = extract_waveform_from_skeleton(skeleton,\n                                         method=\"median\",\n                                         interpolate=True,\n                                         smooth=True,\n                                         smooth_window=7,\n                                         spike_threshold=20)\n\n# 4) save CSV (x, y)\ndf_out = pd.DataFrame({\"x\": np.arange(len(y_series)), \"y\": y_series})\ndf_out.to_csv(out_csv, index=False)\nprint(f\"Saved CSV: {out_csv}\")\n\n# 5) quick visual checks\nplt.figure(figsize=(14,4))\nplt.subplot(1,3,1)\nplt.imshow(cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)); plt.title(\"Original\"); plt.axis('off')\nplt.subplot(1,3,2)\nplt.imshow(binary, cmap='gray'); plt.title(\"Binary (white=wave)\"); plt.axis('off')\nplt.subplot(1,3,3)\nplt.imshow(skeleton, cmap='gray'); plt.title(\"Skeleton (1px)\"); plt.axis('off')\nplt.tight_layout(); plt.show()\n\nplt.figure(figsize=(14,3))\nplt.plot(y_series.values, linewidth=1)\nplt.gca().invert_yaxis()   # optional: invert for typical ECG orientation\nplt.title(\"Extracted ECG waveform (y vs x pixels)\")\nplt.xlabel(\"x (pixels)\")\nplt.ylabel(\"y (pixels)\")\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:26:44.010944Z","iopub.execute_input":"2025-11-16T10:26:44.012025Z","iopub.status.idle":"2025-11-16T10:26:45.820515Z","shell.execute_reply.started":"2025-11-16T10:26:44.011990Z","shell.execute_reply":"2025-11-16T10:26:45.819455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"h, w = skeleton.shape\nlead_y_start = int(h * 0.70)\nlead_y_end = int(h * 0.80)\nlead_ii = skeleton[lead_y_start:lead_y_end, :]\nplt.imshow(lead_ii, cmap='gray')\nplt.title(\"Extracted Lead II Region\")\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:27:26.973467Z","iopub.execute_input":"2025-11-16T10:27:26.973884Z","iopub.status.idle":"2025-11-16T10:27:27.066005Z","shell.execute_reply.started":"2025-11-16T10:27:26.973855Z","shell.execute_reply":"2025-11-16T10:27:27.065047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.signal import savgol_filter\nimport matplotlib.pyplot as plt\n\n# --- Extract waveform (median of y-pixels per column) ---\ny_positions = []\nfor x in range(lead_ii.shape[1]):\n    ys = np.where(lead_ii[:, x] > 0)[0]\n    if len(ys) > 0:\n        y_positions.append(np.median(ys))\n    else:\n        y_positions.append(np.nan)\n\n# --- Build y-series & fill missing values ---\ny_series = pd.Series(y_positions)\n\n# Interpolate → backward-fill → forward-fill\ny_series = (\n    y_series\n    .interpolate()\n    .bfill()\n    .ffill()\n)\n\n# --- Smooth (optional) ---\ny_smooth = savgol_filter(y_series, window_length=31, polyorder=3)\n\n# --- Plot ---\nplt.figure(figsize=(12, 4))\nplt.plot(y_smooth)\nplt.gca().invert_yaxis()\nplt.title(\"Smoothed Single Lead ECG (Lead II)\")\nplt.xlabel(\"Time (pixels)\")\nplt.ylabel(\"Amplitude (pixels)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:28:52.560745Z","iopub.execute_input":"2025-11-16T10:28:52.561264Z","iopub.status.idle":"2025-11-16T10:28:52.787476Z","shell.execute_reply.started":"2025-11-16T10:28:52.561238Z","shell.execute_reply":"2025-11-16T10:28:52.786428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Diagnostic runner: auto-detect skeleton, run extraction, show everything\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\n\n# ---- list of variable names to try ----\ncandidates = [\"skeleton\", \"thin\", \"thin_u8\", \"wave\", \"final\", \"sk\", \"skel\"]\n\nskeleton_var = None\nfor name in candidates:\n    if name in globals():\n        val = globals()[name]\n        if isinstance(val, np.ndarray) and val.ndim == 2:\n            skeleton_var = (name, val)\n            break\n\nif skeleton_var is None:\n    # helpful printable fallback: show available 2D ndarrays\n    print(\"No skeleton-like variable found. Available 2D ndarray variables in notebook:\")\n    found = False\n    for k, v in globals().items():\n        try:\n            import numpy as _np\n            if isinstance(v, _np.ndarray) and v.ndim == 2:\n                print(\"  -\", k, \"shape=\", v.shape)\n                found = True\n        except Exception:\n            pass\n    if not found:\n        print(\"  (none found) — make sure you ran the preprocessing cell that creates 'skeleton' or 'thin'.\")\n    raise NameError(\"No skeleton-like variable found. See printed list above.\")\n\nname_used, skeleton = skeleton_var\nprint(f\"Using skeleton variable: '{name_used}'  shape={skeleton.shape}  dtype={skeleton.dtype}\")\nprint(\"Nonzero pixels in skeleton:\", int((skeleton > 0).sum()))\n\n# show whole skeleton\nplt.figure(figsize=(10,4))\nplt.imshow(skeleton, cmap='gray')\nplt.title(f\"Skeleton: {name_used} (shape={skeleton.shape})\")\nplt.axis('off')\nplt.show()\n\n# Show the lead slice used for extraction (same defaults as the function)\nH, W = skeleton.shape\nlead_y_start = int(H * 0.70)\nlead_y_end   = int(H * 0.78)\nprint(f\"Lead II slice rows: {lead_y_start} to {lead_y_end} (height {lead_y_end-lead_y_start}px)\")\n\nlead_ii = skeleton[lead_y_start:lead_y_end, :]\nplt.figure(figsize=(12,3))\nplt.imshow(lead_ii, cmap='gray')\nplt.title(\"Lead II slice (skeleton region)\")\nplt.axis('off')\nplt.show()\n\n# --- Run the extraction function you have in the notebook ---\n# If you used the function from my earlier message, it's named `extract_and_plot_lead_from_skeleton`.\n# If not present, fallback to a minimal extractor here.\n\nif \"extract_and_plot_lead_from_skeleton\" in globals():\n    func = globals()[\"extract_and_plot_lead_from_skeleton\"]\n    # call function and capture series\n    y_series = func(skeleton, lead_frac=(0.70, 0.78), plot=True)\nelse:\n    # fallback implementation (simple, non-plotted)\n    print(\"Function `extract_and_plot_lead_from_skeleton` not found in globals — using fallback extractor.\")\n    import pandas as _pd\n    y_positions = []\n    for x in range(W):\n        ys = np.where(lead_ii[:, x] > 0)[0]\n        if len(ys) > 0:\n            y_positions.append(float(np.median(ys) + lead_y_start))\n        else:\n            y_positions.append(np.nan)\n    y_series = _pd.Series(y_positions).interpolate().bfill().ffill()\n    # quick smoothing\n    try:\n        from scipy.signal import savgol_filter\n        y_series = pd.Series(savgol_filter(y_series.values, window_length=31 if W>31 else (W//2*2+1), polyorder=3), index=y_series.index)\n    except Exception:\n        pass\n    # plot fallback\n    plt.figure(figsize=(12,3))\n    plt.plot(-y_series.values, color='steelblue')\n    plt.title(\"Fallback extracted waveform (inverted for display)\")\n    plt.axis('on')\n    plt.show()\n\n# --- Show small preview of the numeric data ---\nprint(\"Result: first 10 y-values (pixel coordinates, top=0):\")\nprint(y_series[:10].round(2).to_list())\n\n# Save CSV\nout_csv = \"ecg_waveform_extracted_debug.csv\"\npd.DataFrame({\"x\": np.arange(len(y_series)), \"y\": y_series}).to_csv(out_csv, index=False)\nprint(\"Saved CSV:\", out_csv)\n\n# Return series into notebook namespace for further use\nskeleton_extracted_series = y_series\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:32:44.720286Z","iopub.execute_input":"2025-11-16T10:32:44.720756Z","iopub.status.idle":"2025-11-16T10:32:45.381334Z","shell.execute_reply.started":"2025-11-16T10:32:44.720720Z","shell.execute_reply":"2025-11-16T10:32:45.380064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12,4))\nplt.plot(y_series.values)\nplt.gca().invert_yaxis()\nplt.title(\"Digitized ECG waveform\")\nplt.xlabel(\"Time (pixels)\")\nplt.ylabel(\"Amplitude (pixels)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:33:02.153566Z","iopub.execute_input":"2025-11-16T10:33:02.154022Z","iopub.status.idle":"2025-11-16T10:33:02.323056Z","shell.execute_reply.started":"2025-11-16T10:33:02.153994Z","shell.execute_reply":"2025-11-16T10:33:02.321874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# --- STEP 1: auto-detect a skeleton-like variable ---\ncandidates = [\"wave\", \"skeleton\", \"thin\", \"thin_u8\", \"final\", \"sk\", \"skel\"]\n\nskeleton_img = None\nname_used = None\n\nfor name in candidates:\n    if name in globals():\n        val = globals()[name]\n        if isinstance(val, np.ndarray) and val.ndim == 2:\n            skeleton_img = val\n            name_used = name\n            break\n\nif skeleton_img is None:\n    print(\"❌ No skeleton-like variable found.\")\n    print(\"Here are the 2D arrays currently in memory:\")\n    for k, v in globals().items():\n        if isinstance(v, np.ndarray) and v.ndim == 2:\n            print(\"   →\", k, v.shape, v.dtype)\n    raise NameError(\"You must define a 2D skeleton image first (extracted ECG trace).\")\n\nprint(f\"✔ Using variable '{name_used}' with shape {skeleton_img.shape}\")\n\n# --- STEP 2: extract waveform per column ---\nh, w = skeleton_img.shape\nys = []\n\nfor x in range(w):\n    ys_in_col = np.where(skeleton_img[:, x] > 0)[0]\n    if len(ys_in_col) > 0:\n        ys.append(np.mean(ys_in_col))   # use mean; use median if preferred\n    else:\n        ys.append(np.nan)\n\nys = pd.Series(ys)\n\n# --- STEP 3: interpolate missing values (modern syntax) ---\nys = ys.interpolate().bfill().ffill()\n\n# --- STEP 4: plot ---\nplt.figure(figsize=(12,4))\nplt.plot(ys.values)\nplt.gca().invert_yaxis()\nplt.title(f\"Digitized ECG waveform (from '{name_used}')\")\nplt.xlabel(\"Time (pixels)\")\nplt.ylabel(\"Amplitude (pixels)\")\nplt.grid(True, alpha=0.3)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:34:19.131342Z","iopub.execute_input":"2025-11-16T10:34:19.131616Z","iopub.status.idle":"2025-11-16T10:34:19.368440Z","shell.execute_reply.started":"2025-11-16T10:34:19.131598Z","shell.execute_reply":"2025-11-16T10:34:19.367442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time = np.arange(len(ys)) / 500  # 仮の500Hzサンプリング\namp = (ys - ys.mean()) / ys.std()\n\ndf_wave = pd.DataFrame({\"time\": time, \"amplitude\": amp})\nout_path = \"/kaggle/working/sample_waveform.csv\"\ndf_wave.to_csv(out_path, index=False)\nprint(f\"Saved → {out_path}\")\ndf_wave.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:34:36.411342Z","iopub.execute_input":"2025-11-16T10:34:36.411642Z","iopub.status.idle":"2025-11-16T10:34:36.446895Z","shell.execute_reply.started":"2025-11-16T10:34:36.411619Z","shell.execute_reply":"2025-11-16T10:34:36.446097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====== DIAGNOSTIC CELL: always produces visible output ======\n\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# 1. Check if skeleton exists\nif \"skeleton\" not in globals():\n    raise NameError(\"❌ 'skeleton' variable does not exist. Preprocessing did not run.\")\n\nprint(\"✔ skeleton exists. Shape =\", skeleton.shape)\n\n# 2. Show skeleton image\nplt.figure(figsize=(12,4))\nplt.imshow(skeleton, cmap='gray')\nplt.title(\"Skeleton Image\")\nplt.axis('off')\nplt.show()\n\nprint(\"Nonzero skeleton pixels:\", int((skeleton > 0).sum()))\n\n# 3. Projection plot\nproj = np.sum(skeleton > 0, axis=1)\n\nplt.figure(figsize=(12,3))\nplt.plot(proj)\nplt.title(\"Row Projection (used for Lead detection)\")\nplt.xlabel(\"Row (y)\")\nplt.ylabel(\"White pixel count\")\nplt.grid(True)\nplt.show()\n\n# 4. Try running extractor with plot=True\nprint(\"\\nRunning extractor...\\n\")\ny = extract_lead_ii_from_skeleton(skeleton, plot=True)\n\n# 5. Show first values\nprint(\"\\nFirst 10 extracted y-values:\\n\", y.head(10).round(2).to_list())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:37:30.796907Z","iopub.execute_input":"2025-11-16T10:37:30.797259Z","iopub.status.idle":"2025-11-16T10:37:31.625150Z","shell.execute_reply.started":"2025-11-16T10:37:30.797236Z","shell.execute_reply":"2025-11-16T10:37:31.624038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T10:44:53.546426Z","iopub.execute_input":"2025-11-16T10:44:53.547120Z","iopub.status.idle":"2025-11-16T10:44:53.589669Z","shell.execute_reply.started":"2025-11-16T10:44:53.547095Z","shell.execute_reply":"2025-11-16T10:44:53.588989Z"}},"outputs":[],"execution_count":null}]}