{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:22:24.120982Z","iopub.execute_input":"2026-01-03T17:22:24.121394Z","iopub.status.idle":"2026-01-03T17:22:24.550720Z","shell.execute_reply.started":"2026-01-03T17:22:24.121361Z","shell.execute_reply":"2026-01-03T17:22:24.549637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/physionet-ecg-image-digitization\"\nTRAIN_PATH = f\"{BASE_PATH}/train\"\nTEST_PATH  = f\"{BASE_PATH}/test\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:22:38.230097Z","iopub.execute_input":"2026-01-03T17:22:38.230555Z","iopub.status.idle":"2026-01-03T17:22:38.235265Z","shell.execute_reply.started":"2026-01-03T17:22:38.230522Z","shell.execute_reply":"2026-01-03T17:22:38.234287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_meta = pd.read_csv(f\"{BASE_PATH}/train.csv\")\ntest_meta  = pd.read_csv(f\"{BASE_PATH}/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:22:46.867724Z","iopub.execute_input":"2026-01-03T17:22:46.868661Z","iopub.status.idle":"2026-01-03T17:22:46.879688Z","shell.execute_reply.started":"2026-01-03T17:22:46.868625Z","shell.execute_reply":"2026-01-03T17:22:46.878662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LEAD_ORDER = [\n    [\"I\",   \"aVR\", \"V1\", \"V4\"],\n    [\"II\",  \"aVL\", \"V2\", \"V5\"],\n    [\"III\", \"aVF\", \"V3\", \"V6\"]\n]\n\nLEAD_TO_POS = {}\nfor r, row in enumerate(LEAD_ORDER):\n    for c, lead in enumerate(row):\n        LEAD_TO_POS[lead] = (r, c)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:23:05.884960Z","iopub.execute_input":"2026-01-03T17:23:05.885322Z","iopub.status.idle":"2026-01-03T17:23:05.891687Z","shell.execute_reply.started":"2026-01-03T17:23:05.885293Z","shell.execute_reply":"2026-01-03T17:23:05.890460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_image(path):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    img = cv2.normalize(img, None, 0, 255, cv2.NORM_MINMAX)\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:23:16.085493Z","iopub.execute_input":"2026-01-03T17:23:16.085789Z","iopub.status.idle":"2026-01-03T17:23:16.090686Z","shell.execute_reply.started":"2026-01-03T17:23:16.085764Z","shell.execute_reply":"2026-01-03T17:23:16.089747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def crop_lead(img, lead):\n    h, w = img.shape\n    rows, cols = 3, 4\n\n    r, c = LEAD_TO_POS[lead]\n\n    y1 = int(r * h / rows)\n    y2 = int((r + 1) * h / rows)\n    x1 = int(c * w / cols)\n    x2 = int((c + 1) * w / cols)\n\n    return img[y1:y2, x1:x2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:23:25.601113Z","iopub.execute_input":"2026-01-03T17:23:25.601441Z","iopub.status.idle":"2026-01-03T17:23:25.607372Z","shell.execute_reply.started":"2026-01-03T17:23:25.601413Z","shell.execute_reply":"2026-01-03T17:23:25.606480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_signal(lead_img, out_len):\n    # Adaptive threshold\n    bw = cv2.adaptiveThreshold(\n        lead_img, 255,\n        cv2.ADAPTIVE_THRESH_GAUSSIAN_C,\n        cv2.THRESH_BINARY_INV, 31, 5\n    )\n\n    h, w = bw.shape\n\n    # Column-wise centerline\n    y = np.zeros(w)\n    for i in range(w):\n        col = bw[:, i]\n        idx = np.where(col > 0)[0]\n        y[i] = idx.mean() if len(idx) else h / 2\n\n    # Normalize (pixel → relative amplitude)\n    y = (h - y) / h\n\n    # Resample to required length\n    x_old = np.linspace(0, 1, len(y))\n    x_new = np.linspace(0, 1, out_len)\n\n    signal = np.interp(x_new, x_old, y)\n\n    # Zero-mean (metric removes offset anyway)\n    signal -= signal.mean()\n\n    return signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:23:44.166275Z","iopub.execute_input":"2026-01-03T17:23:44.166669Z","iopub.status.idle":"2026-01-03T17:23:44.174218Z","shell.execute_reply.started":"2026-01-03T17:23:44.166638Z","shell.execute_reply":"2026-01-03T17:23:44.173100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_training_pairs(limit_ids=None):\n    samples = []\n\n    ids = train_meta[\"id\"].tolist()\n    if limit_ids:\n        ids = ids[:limit_ids]\n\n    for ecg_id in tqdm(ids):\n        folder = f\"{TRAIN_PATH}/{ecg_id}\"\n        ts = pd.read_csv(f\"{folder}/{ecg_id}.csv\")\n\n        image_paths = glob(f\"{folder}/{ecg_id}-*.png\")\n\n        for img_path in image_paths:\n            img = load_image(img_path)\n\n            for lead in ts.columns:\n                lead_img = crop_lead(img, lead)\n                signal = ts[lead].values.astype(np.float32)\n\n                samples.append({\n                    \"image\": lead_img,\n                    \"signal\": signal,\n                    \"lead\": lead\n                })\n\n    return samples","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:23:58.845761Z","iopub.execute_input":"2026-01-03T17:23:58.846163Z","iopub.status.idle":"2026-01-03T17:23:58.852786Z","shell.execute_reply.started":"2026-01-03T17:23:58.846130Z","shell.execute_reply":"2026-01-03T17:23:58.851979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_test():\n    preds = []\n\n    for _, row in tqdm(test_meta.iterrows(), total=len(test_meta)):\n        img_path = f\"{TEST_PATH}/{row.id}.png\"\n        img = load_image(img_path)\n\n        signal = extract_signal(\n            img,\n            out_len=row.number_of_rows\n        )\n\n        for i, v in enumerate(signal):\n            preds.append({\n                \"id\": f\"{row.id}_{i}_{row.lead}\",\n                \"value\": float(v)\n            })\n\n    return pd.DataFrame(preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:24:17.720617Z","iopub.execute_input":"2026-01-03T17:24:17.720945Z","iopub.status.idle":"2026-01-03T17:24:17.727745Z","shell.execute_reply.started":"2026-01-03T17:24:17.720919Z","shell.execute_reply":"2026-01-03T17:24:17.726590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = predict_test()\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-03T17:24:28.328961Z","iopub.execute_input":"2026-01-03T17:24:28.329324Z","iopub.status.idle":"2026-01-03T17:24:33.117294Z","shell.execute_reply.started":"2026-01-03T17:24:28.329294Z","shell.execute_reply":"2026-01-03T17:24:33.116279Z"}},"outputs":[],"execution_count":null}]}