{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:18.554Z","iopub.execute_input":"2025-11-14T15:50:18.555245Z","iopub.status.idle":"2025-11-14T15:50:19.956528Z","shell.execute_reply.started":"2025-11-14T15:50:18.555203Z","shell.execute_reply":"2025-11-14T15:50:19.954836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#powershell\n\necg-digitizer/\n├── data/\n│   ├── raw/                # put example images here (eg: eg1.png)\n│   └── out/                # all output files: CSVs, extracted traces, debug images\n├── src/\n│   ├── __init__.py\n│   ├── preprocessing.py    # image cleanup, deskewing, denoising\n│   ├── grid_detection.py   # detect ECG grid lines, spacing, scaling\n│   ├── lead_detection.py   # crop 12-lead regions, detect bounding boxes\n│   ├── classical_trace.py  # extract waveform pixels → y-values\n│   ├── convert.py          # convert pixel waveform → time-series signals\n│   └── inference.py        # full pipeline: image → CSV, orchestrates everything\n├── quick_probe.py          # testing script (rotation, scale, variations)\n├── requirements.txt        # all Python dependencies\n└── README.md               # how to install + how to run the tool\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:19.958342Z","iopub.execute_input":"2025-11-14T15:50:19.958641Z","iopub.status.idle":"2025-11-14T15:50:19.966049Z","shell.execute_reply.started":"2025-11-14T15:50:19.9586Z","shell.execute_reply":"2025-11-14T15:50:19.964657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#src/preprocessing.py\n\nimport cv2\nimport numpy as np\n\ndef load_img(path):\n    im = cv2.imread(path, cv2.IMREAD_COLOR)\n    if im is None:\n        raise FileNotFoundError(f\"Image not found: {path}\")\n    return im\n\ndef normalize_img(im, max_side=2000):\n    h,w = im.shape[:2]\n    scale = max_side / max(h,w)\n    if scale < 1.0:\n        im = cv2.resize(im, (int(w*scale), int(h*scale)), interpolation=cv2.INTER_AREA)\n    return im\n\ndef to_gray(im):\n    return cv2.cvtColor(im, cv2.COLOR_BGR2GRAY)\n\ndef auto_deskew(gray):\n    # simple deskew using largest contour rotated bounding box\n    _, th = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    coords = cv2.findNonZero(255 - th)  # find ink pixels\n    if coords is None:\n        return gray, 0.0\n    rect = cv2.minAreaRect(coords)\n    angle = rect[-1]\n    if angle < -45:\n        angle = -(90 + angle)\n    else:\n        angle = -angle\n    (h,w) = gray.shape\n    M = cv2.getRotationMatrix2D((w/2,h/2), angle, 1.0)\n    rotated = cv2.warpAffine(gray, M, (w,h), flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE)\n    return rotated, angle\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:19.966529Z","iopub.status.idle":"2025-11-14T15:50:19.966893Z","shell.execute_reply.started":"2025-11-14T15:50:19.966717Z","shell.execute_reply":"2025-11-14T15:50:19.96674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# src/grid_detection.py\nimport numpy as np\nfrom scipy.signal import find_peaks\n\ndef estimate_grid_spacing(gray, debug=False):\n    \"\"\"\n    Estimate vertical grid spacing (pixels per small grid cell) by summing columns\n    and finding peaks where grid lines appear dark.\n    \"\"\"\n    col_sum = gray.mean(axis=0)\n    # invert so grid dark lines become peaks\n    signal = -col_sum\n    # smooth a little\n    from scipy.ndimage import gaussian_filter1d\n    s = gaussian_filter1d(signal, sigma=3)\n    peaks, props = find_peaks(s, distance=5, prominence=np.std(s)*0.5)\n    if len(peaks) < 2:\n        return None\n    spacings = np.diff(peaks)\n    median_spacing = int(np.median(spacings))\n    if debug:\n        return median_spacing, peaks\n    return median_spacing\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:19.9682Z","iopub.status.idle":"2025-11-14T15:50:19.96861Z","shell.execute_reply.started":"2025-11-14T15:50:19.968435Z","shell.execute_reply":"2025-11-14T15:50:19.968452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# src/lead_detection.py\nimport cv2\nimport numpy as np\n\nSTANDARD_LEADS = [\"I\",\"II\",\"III\",\"aVR\",\"aVL\",\"aVF\",\"V1\",\"V2\",\"V3\",\"V4\",\"V5\",\"V6\"]\n\ndef detect_leads_grid(gray):\n    \"\"\"\n    Heuristic: assume ECG contains 3 rows x 4 columns of lead panels.\n    Return dict lead_name -> crop_image (grayscale).\n    \"\"\"\n    h,w = gray.shape\n    rows = 3\n    cols = 4\n    pad_h = int(0.02 * h)\n    pad_w = int(0.02 * w)\n    cell_h = h // rows\n    cell_w = w // cols\n    results = {}\n    idx = 0\n    for r in range(rows):\n        for c in range(cols):\n            y0 = max(0, r*cell_h - pad_h)\n            y1 = min(h, (r+1)*cell_h + pad_h)\n            x0 = max(0, c*cell_w - pad_w)\n            x1 = min(w, (c+1)*cell_w + pad_w)\n            crop = gray[y0:y1, x0:x1].copy()\n            if idx < len(STANDARD_LEADS):\n                results[STANDARD_LEADS[idx]] = crop\n            idx += 1\n    return results\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:19.969929Z","iopub.status.idle":"2025-11-14T15:50:19.970289Z","shell.execute_reply.started":"2025-11-14T15:50:19.970127Z","shell.execute_reply":"2025-11-14T15:50:19.970144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# src/classical_trace.py\nimport numpy as np\nimport cv2\nimport pandas as pd\n\ndef mask_from_crop(crop):\n    # adaptive threshold + morphological cleaning\n    blur = cv2.GaussianBlur(crop, (5,5), 0)\n    th = cv2.adaptiveThreshold(blur, 255, cv2.ADAPTIVE_THRESH_MEAN_C,\n                                cv2.THRESH_BINARY_INV, 15, 8)\n    # remove grid using morphological open to reduce horizontal/vertical lines\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (3,3))\n    cleaned = cv2.morphologyEx(th, cv2.MORPH_OPEN, kernel, iterations=1)\n    return cleaned\n\ndef centerline_from_mask(mask):\n    \"\"\"\n    For each column, compute the mean y of mask pixels (trace pixels == 255).\n    Interpolate missing columns.\n    Returns numpy array length==width with y pixel positions (float).\n    \"\"\"\n    h,w = mask.shape\n    trace = np.full(w, np.nan, dtype=np.float32)\n    for x in range(w):\n        ys = np.where(mask[:,x] > 0)[0]\n        if ys.size:\n            trace[x] = ys.mean()\n    # interpolate\n    s = pd.Series(trace)\n    trace_filled = s.interpolate(limit_direction='both').ffill().bfill().values\n    return trace_filled\n\ndef trace_pipeline(crop):\n    mask = mask_from_crop(crop)\n    center = centerline_from_mask(mask)\n    return center, mask\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:19.971719Z","iopub.status.idle":"2025-11-14T15:50:19.972255Z","shell.execute_reply.started":"2025-11-14T15:50:19.972117Z","shell.execute_reply":"2025-11-14T15:50:19.972131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# src/convert.py\nimport numpy as np\n\ndef pixel_to_mv_time(centerline_pixels, px_per_mm, fs=500, lead_duration_s=2.5):\n    \"\"\"\n    Convert centerline (y-pixels per x-pixel) to voltage (mV) time-series.\n    Assumptions:\n      - 1 mV = 10 mm vertically (standard)\n      - px_per_mm is pixels per mm (grid spacing)\n      - centerline_pixels array length = width (pixels). We'll map width -> number of samples = floor(fs * lead_duration_s)\n    Returns: time_series (numpy array of length floor(fs*lead_duration_s))\n    \"\"\"\n    if px_per_mm is None or px_per_mm <= 0:\n        raise ValueError(\"px_per_mm must be positive\")\n    mm_per_pixel = 1.0 / px_per_mm\n    # convert pixel y -> mm displacement from midline (we don't know midline; centerline is in pixels from top)\n    # to convert to mV we need a reference baseline; assume center of crop is baseline\n    h = 1  # not used here directly; we'll compute relative to mean\n    y = np.array(centerline_pixels, dtype=np.float32)\n    y = y - np.mean(y)  # remove baseline (center)\n    # mm displacement:\n    y_mm = y * mm_per_pixel\n    # mV: 1 mV = 10 mm vertically\n    y_mv = y_mm / 10.0\n    # resample/interpolate to desired number of samples\n    num_samples = int(np.floor(fs * lead_duration_s))\n    if num_samples <= 0:\n        raise ValueError(\"num_samples must be >0\")\n    # original x positions\n    orig_x = np.linspace(0, 1, len(y_mv))\n    target_x = np.linspace(0, 1, num_samples)\n    interp = np.interp(target_x, orig_x, y_mv)\n    return interp\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:19.973686Z","iopub.status.idle":"2025-11-14T15:50:19.974046Z","shell.execute_reply.started":"2025-11-14T15:50:19.97387Z","shell.execute_reply":"2025-11-14T15:50:19.973886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# src/inference.py\nimport os\nfrom .preprocessing import load_img, normalize_img, to_gray, auto_deskew\nfrom .grid_detection import estimate_grid_spacing\nfrom .lead_detection import detect_leads_grid\nfrom .classical_trace import trace_pipeline\nfrom .convert import pixel_to_mv_time\nimport pandas as pd\n\ndef process_image(path, out_dir, fs=500, lead_durations=None):\n    im = load_img(path)\n    im = normalize_img(im, max_side=1600)\n    gray = to_gray(im)\n    gray, angle = auto_deskew(gray)\n    px_per_mm = estimate_grid_spacing(gray)\n    if px_per_mm is None:\n        # fallback: assume 3 px per mm (low-res)\n        px_per_mm = 3\n    leads = detect_leads_grid(gray)\n    if lead_durations is None:\n        # typical: lead II = 10s, others 2.5s, but for dataset you may pass fs and durations\n        lead_durations = {k: (10.0 if k==\"II\" else 2.5) for k in leads.keys()}\n    os.makedirs(out_dir, exist_ok=True)\n    results = {}\n    for name, crop in leads.items():\n        center, mask = trace_pipeline(crop)\n        duration = lead_durations.get(name, 2.5)\n        signal = pixel_to_mv_time(center, px_per_mm, fs=fs, lead_duration_s=duration)\n        # save CSV\n        df = pd.DataFrame({\"value\": signal})\n        fname = os.path.join(out_dir, f\"{os.path.basename(path).split('.')[0]}_{name}.csv\")\n        df.to_csv(fname, index_label=\"row_id\")\n        results[name] = {\"csv\": fname, \"signal\": signal, \"mask\": mask}\n    return results\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:25.59253Z","iopub.execute_input":"2025-11-14T15:50:25.594069Z","iopub.status.idle":"2025-11-14T15:50:25.693245Z","shell.execute_reply.started":"2025-11-14T15:50:25.59403Z","shell.execute_reply":"2025-11-14T15:50:25.691711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# quick_probe.py\nimport sys\nfrom src.inference import process_image\nimport argparse\nimport os\n\nparser = argparse.ArgumentParser()\nparser.add_argument(\"image\", help=\"path to ECG image\")\nparser.add_argument(\"--out\", default=\"data/out\", help=\"output dir\")\nparser.add_argument(\"--fs\", type=int, default=500, help=\"sampling freq (Hz)\")\nargs = parser.parse_args()\n\nres = process_image(args.image, args.out, fs=args.fs)\nprint(\"Done. Outputs:\")\nfor lead, info in res.items():\n    print(f\"  {lead}: {info['csv']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:27.153043Z","iopub.execute_input":"2025-11-14T15:50:27.153414Z","iopub.status.idle":"2025-11-14T15:50:27.168935Z","shell.execute_reply.started":"2025-11-14T15:50:27.153391Z","shell.execute_reply":"2025-11-14T15:50:27.167403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#README.md\n# ECG Digitizer (minimal classical pipeline)\n\n1. Create and activate venv:\n   python -m venv .venv\n   # Windows:\n   .venv\\Scripts\\activate\n   # macOS/Linux:\n   source .venv/bin/activate\n\n2. Install dependencies:\n   pip install -r requirements.txt\n\n3. Place an example ECG image in `data/raw/eg1.png`\n\n4. Run:\n   python quick_probe.py data/raw/eg1.png --out data/out --fs 500\n\n5. Outputs (per lead) CSVs saved in data/out/*.csv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:27.169607Z","iopub.status.idle":"2025-11-14T15:50:27.169971Z","shell.execute_reply.started":"2025-11-14T15:50:27.169822Z","shell.execute_reply":"2025-11-14T15:50:27.169836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#bash/powershell\n\n1.python quick_probe.py data/raw/eg1.png\n\n2.python quick_probe.py data/raw/kaggle/input/physionet-ecg-image-digitization/train/1678324396/1678324396-0001.png","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T15:50:27.171768Z","iopub.status.idle":"2025-11-14T15:50:27.172218Z","shell.execute_reply.started":"2025-11-14T15:50:27.171966Z","shell.execute_reply":"2025-11-14T15:50:27.171982Z"}},"outputs":[],"execution_count":null}]}