{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom tqdm import tqdm\nfrom scipy.signal import savgol_filter, butter, filtfilt, correlate\nfrom scipy.interpolate import interp1d\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T12:59:47.484525Z","iopub.execute_input":"2026-01-11T12:59:47.485370Z","iopub.status.idle":"2026-01-11T12:59:47.490502Z","shell.execute_reply.started":"2026-01-11T12:59:47.485324Z","shell.execute_reply":"2026-01-11T12:59:47.489537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/physionet-ecg-image-digitization\"\n\ntrain_meta = pd.read_csv(f\"{DATA_DIR}/train.csv\")\ntest_meta  = pd.read_csv(f\"{DATA_DIR}/test.csv\")\n\nprint(train_meta.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T12:59:49.960768Z","iopub.execute_input":"2026-01-11T12:59:49.961396Z","iopub.status.idle":"2026-01-11T12:59:49.984331Z","shell.execute_reply.started":"2026-01-11T12:59:49.961364Z","shell.execute_reply":"2026-01-11T12:59:49.983080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nBASE_DIR = \"/kaggle/input\"\n\nfor d in os.listdir(BASE_DIR):\n    print(d)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T12:59:55.183967Z","iopub.execute_input":"2026-01-11T12:59:55.184952Z","iopub.status.idle":"2026-01-11T12:59:55.190349Z","shell.execute_reply.started":"2026-01-11T12:59:55.184908Z","shell.execute_reply":"2026-01-11T12:59:55.189476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/physionet-ecg-image-digitization\"\n\nfor d in os.listdir(DATA_DIR):\n    print(d)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T12:59:57.902297Z","iopub.execute_input":"2026-01-11T12:59:57.903087Z","iopub.status.idle":"2026-01-11T12:59:57.910539Z","shell.execute_reply.started":"2026-01-11T12:59:57.903047Z","shell.execute_reply":"2026-01-11T12:59:57.909680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATA_DIR = \"/kaggle/input/physionet-ecg-image-digitization\"\n\nprint(\"Train images:\", os.listdir(f\"{DATA_DIR}/train\")[:5])\nprint(\"Test images:\", os.listdir(f\"{DATA_DIR}/test\")[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:00.771488Z","iopub.execute_input":"2026-01-11T13:00:00.771839Z","iopub.status.idle":"2026-01-11T13:00:00.779906Z","shell.execute_reply.started":"2026-01-11T13:00:00.771809Z","shell.execute_reply":"2026-01-11T13:00:00.778884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image(path):\n    import cv2, numpy as np, os\n    if not os.path.exists(path):\n        raise FileNotFoundError(path)\n\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        raise ValueError(f\"Failed to load {path}\")\n\n    img = cv2.resize(img, (2048, 1024))\n    img = cv2.equalizeHist(img)\n    return img.astype(np.float32) / 255.0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:03.716131Z","iopub.execute_input":"2026-01-11T13:00:03.716460Z","iopub.status.idle":"2026-01-11T13:00:03.722868Z","shell.execute_reply.started":"2026-01-11T13:00:03.716420Z","shell.execute_reply":"2026-01-11T13:00:03.721747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nDATA_DIR = \"/kaggle/input/physionet-ecg-image-digitization\"\n\ntest_meta = pd.read_csv(f\"{DATA_DIR}/test.csv\")\nprint(test_meta.columns)\nprint(test_meta.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:07.554263Z","iopub.execute_input":"2026-01-11T13:00:07.554606Z","iopub.status.idle":"2026-01-11T13:00:07.566707Z","shell.execute_reply.started":"2026-01-11T13:00:07.554574Z","shell.execute_reply":"2026-01-11T13:00:07.565774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom scipy.signal import savgol_filter, butter, filtfilt\nfrom scipy.interpolate import interp1d\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:10.943926Z","iopub.execute_input":"2026-01-11T13:00:10.944974Z","iopub.status.idle":"2026-01-11T13:00:10.949675Z","shell.execute_reply.started":"2026-01-11T13:00:10.944928Z","shell.execute_reply":"2026-01-11T13:00:10.948682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/physionet-ecg-image-digitization\"\n\ntrain_meta = pd.read_csv(f\"{DATA_DIR}/train.csv\")\ntest_meta  = pd.read_csv(f\"{DATA_DIR}/test.csv\")\n\nprint(test_meta.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:14.122570Z","iopub.execute_input":"2026-01-11T13:00:14.122980Z","iopub.status.idle":"2026-01-11T13:00:14.138226Z","shell.execute_reply.started":"2026-01-11T13:00:14.122950Z","shell.execute_reply":"2026-01-11T13:00:14.137243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_image(ecg_id, split=\"test\"):\n    path = f\"{DATA_DIR}/{split}/{ecg_id}.png\"\n    if not os.path.exists(path):\n        raise FileNotFoundError(path)\n\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        raise ValueError(f\"Failed to load {path}\")\n\n    img = cv2.resize(img, (2048, 1024))\n    img = cv2.equalizeHist(img)\n    img = img.astype(np.float32) / 255.0\n    return img\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:18.019998Z","iopub.execute_input":"2026-01-11T13:00:18.020942Z","iopub.status.idle":"2026-01-11T13:00:18.026766Z","shell.execute_reply.started":"2026-01-11T13:00:18.020901Z","shell.execute_reply":"2026-01-11T13:00:18.025745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LEAD_ORDER = [\n    'I','II','III','aVR','aVL','aVF',\n    'V1','V2','V3','V4','V5','V6'\n]\n\ndef split_leads(img):\n    h, w = img.shape\n    lh, lw = h // 6, w // 2\n\n    lead_imgs = {}\n    idx = 0\n    for r in range(6):\n        for c in range(2):\n            if idx >= len(LEAD_ORDER):\n                break\n            lead_imgs[LEAD_ORDER[idx]] = img[\n                r*lh:(r+1)*lh,\n                c*lw:(c+1)*lw\n            ]\n            idx += 1\n    return lead_imgs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:23.207018Z","iopub.execute_input":"2026-01-11T13:00:23.207330Z","iopub.status.idle":"2026-01-11T13:00:23.214484Z","shell.execute_reply.started":"2026-01-11T13:00:23.207304Z","shell.execute_reply":"2026-01-11T13:00:23.213559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_waveform(lead_img, target_len):\n    h, w = lead_img.shape\n    y = []\n\n    for x in range(w):\n        col = lead_img[:, x]\n        ys = np.where(col < 0.8)[0]   # dark trace\n        y.append(np.mean(ys) if len(ys) else h//2)\n\n    y = h - np.array(y)\n    y = (y - np.mean(y)) / (np.std(y) + 1e-6)\n\n    f = interp1d(np.linspace(0,1,len(y)), y)\n    return f(np.linspace(0,1,target_len))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:27.731088Z","iopub.execute_input":"2026-01-11T13:00:27.731454Z","iopub.status.idle":"2026-01-11T13:00:27.738413Z","shell.execute_reply.started":"2026-01-11T13:00:27.731423Z","shell.execute_reply":"2026-01-11T13:00:27.737135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def bandpass(sig, fs):\n    b, a = butter(3, [0.5/(fs/2), 40/(fs/2)], btype='band')\n    return filtfilt(b, a, sig)\n\ndef smooth(sig, fs):\n    sig = savgol_filter(sig, 31, 3)\n    return bandpass(sig, fs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:37.736847Z","iopub.execute_input":"2026-01-11T13:00:37.737225Z","iopub.status.idle":"2026-01-11T13:00:37.743009Z","shell.execute_reply.started":"2026-01-11T13:00:37.737193Z","shell.execute_reply":"2026-01-11T13:00:37.741975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = []\n\ngrouped = test_meta.groupby(\"id\")\n\nfor ecg_id, rows in tqdm(grouped):\n    img = load_image(ecg_id, split=\"test\")\n    lead_imgs = split_leads(img)\n\n    for _, row in rows.iterrows():\n        lead = row.lead\n        fs = row.fs\n        sig_len = row.number_of_rows\n\n        lead_img = lead_imgs[lead]\n        signal = extract_waveform(lead_img, sig_len)\n        signal = smooth(signal, fs)\n\n        for i, val in enumerate(signal):\n            submission.append({\n                \"id\": f\"{ecg_id}_{i}_{lead}\",\n                \"value\": float(val)\n            })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:42.439414Z","iopub.execute_input":"2026-01-11T13:00:42.440304Z","iopub.status.idle":"2026-01-11T13:00:43.055485Z","shell.execute_reply.started":"2026-01-11T13:00:42.440265Z","shell.execute_reply":"2026-01-11T13:00:43.054541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame(submission)\nsubmission_df.to_csv(\"submission.csv\", index=False)\n\nsubmission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:46.635033Z","iopub.execute_input":"2026-01-11T13:00:46.635880Z","iopub.status.idle":"2026-01-11T13:00:46.879729Z","shell.execute_reply.started":"2026-01-11T13:00:46.635828Z","shell.execute_reply":"2026-01-11T13:00:46.878872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.signal import correlate\n\ndef align_signal(pred, fs):\n    \"\"\"\n    Aligns signal in time (+/- 0.2s) and removes vertical offset.\n    \"\"\"\n    max_shift = int(0.2 * fs)\n\n    # self-correlation to stabilize alignment\n    corr = correlate(pred, pred, mode=\"full\")\n    shift = np.argmax(corr) - len(pred)\n\n    shift = np.clip(shift, -max_shift, max_shift)\n\n    if shift > 0:\n        pred = pred[shift:]\n    elif shift < 0:\n        pred = pred[:shift]\n\n    # vertical shift removal\n    pred = pred - np.mean(pred)\n\n    return pred\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:53.065242Z","iopub.execute_input":"2026-01-11T13:00:53.065582Z","iopub.status.idle":"2026-01-11T13:00:53.072550Z","shell.execute_reply.started":"2026-01-11T13:00:53.065553Z","shell.execute_reply":"2026-01-11T13:00:53.071513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reconstruct_variants(lead_img, fs, sig_len):\n    variants = []\n\n    # Variant 1: standard smoothing\n    s1 = extract_waveform(lead_img, sig_len)\n    s1 = smooth(s1, fs)\n    variants.append(s1)\n\n    # Variant 2: stronger smoothing\n    s2 = extract_waveform(lead_img, sig_len)\n    s2 = savgol_filter(s2, 51, 3)\n    s2 = bandpass(s2, fs)\n    variants.append(s2)\n\n    # Variant 3: lighter smoothing\n    s3 = extract_waveform(lead_img, sig_len)\n    s3 = savgol_filter(s3, 21, 2)\n    s3 = bandpass(s3, fs)\n    variants.append(s3)\n\n    return variants\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:00:57.361434Z","iopub.execute_input":"2026-01-11T13:00:57.362247Z","iopub.status.idle":"2026-01-11T13:00:57.368314Z","shell.execute_reply.started":"2026-01-11T13:00:57.362207Z","shell.execute_reply":"2026-01-11T13:00:57.367300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ensemble_signals(signals):\n    min_len = min(len(s) for s in signals)\n    signals = [s[:min_len] for s in signals]\n    return np.mean(signals, axis=0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:04.311099Z","iopub.execute_input":"2026-01-11T13:01:04.311455Z","iopub.status.idle":"2026-01-11T13:01:04.317069Z","shell.execute_reply.started":"2026-01-11T13:01:04.311424Z","shell.execute_reply":"2026-01-11T13:01:04.316172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = []\n\ngrouped = test_meta.groupby(\"id\")\n\nfor ecg_id, rows in tqdm(grouped, desc=\"Inference\"):\n    img = load_image(ecg_id, split=\"test\")\n    lead_imgs = split_leads(img)\n\n    for _, row in rows.iterrows():\n        lead = row.lead\n        fs = row.fs\n        sig_len = row.number_of_rows\n\n        lead_img = lead_imgs[lead]\n\n        # ---- ensemble reconstruction ----\n        variants = reconstruct_variants(lead_img, fs, sig_len)\n        signal = ensemble_signals(variants)\n\n        # ---- metric-aware alignment ----\n        signal = align_signal(signal, fs)\n\n        for i, v in enumerate(signal):\n            submission.append({\n                \"id\": f\"{ecg_id}_{i}_{lead}\",\n                \"value\": float(v)\n            })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:09.600484Z","iopub.execute_input":"2026-01-11T13:01:09.601333Z","iopub.status.idle":"2026-01-11T13:01:10.866124Z","shell.execute_reply.started":"2026-01-11T13:01:09.601298Z","shell.execute_reply":"2026-01-11T13:01:10.865245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_parquet(\"submission.parquet\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:14.647011Z","iopub.execute_input":"2026-01-11T13:01:14.647367Z","iopub.status.idle":"2026-01-11T13:01:14.695019Z","shell.execute_reply.started":"2026-01-11T13:01:14.647335Z","shell.execute_reply":"2026-01-11T13:01:14.694123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Expected rows:\", test_meta[\"number_of_rows\"].sum())\nprint(\"Your rows:\", len(submission_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:17.983283Z","iopub.execute_input":"2026-01-11T13:01:17.983659Z","iopub.status.idle":"2026-01-11T13:01:17.990470Z","shell.execute_reply.started":"2026-01-11T13:01:17.983601Z","shell.execute_reply":"2026-01-11T13:01:17.989182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dup = submission_df[\"id\"].duplicated().sum()\nprint(\"Duplicate IDs:\", dup)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:21.474254Z","iopub.execute_input":"2026-01-11T13:01:21.474608Z","iopub.status.idle":"2026-01-11T13:01:21.491481Z","shell.execute_reply.started":"2026-01-11T13:01:21.474579Z","shell.execute_reply":"2026-01-11T13:01:21.490584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expected_ids = []\n\nfor _, row in test_meta.iterrows():\n    for i in range(row.number_of_rows):\n        expected_ids.append(f\"{row.id}_{i}_{row.lead}\")\n\nexpected_ids = set(expected_ids)\nsubmitted_ids = set(submission_df[\"id\"])\n\nprint(\"Missing:\", len(expected_ids - submitted_ids))\nprint(\"Extra:\", len(submitted_ids - expected_ids))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:24.759032Z","iopub.execute_input":"2026-01-11T13:01:24.760179Z","iopub.status.idle":"2026-01-11T13:01:25.474749Z","shell.execute_reply.started":"2026-01-11T13:01:24.760138Z","shell.execute_reply":"2026-01-11T13:01:25.473833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"NaNs:\", submission_df[\"value\"].isna().sum())\nprint(\"Infs:\", np.isinf(submission_df[\"value\"]).sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:28.593517Z","iopub.execute_input":"2026-01-11T13:01:28.593883Z","iopub.status.idle":"2026-01-11T13:01:28.600778Z","shell.execute_reply.started":"2026-01-11T13:01:28.593853Z","shell.execute_reply":"2026-01-11T13:01:28.599839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame(submission)\nsubmission_df.to_csv(\"submission.csv\", index=False)\n\nsubmission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:31.456515Z","iopub.execute_input":"2026-01-11T13:01:31.456887Z","iopub.status.idle":"2026-01-11T13:01:31.698955Z","shell.execute_reply.started":"2026-01-11T13:01:31.456857Z","shell.execute_reply":"2026-01-11T13:01:31.698144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nsample = pd.read_parquet(\n    \"/kaggle/input/physionet-ecg-image-digitization/sample_submission.parquet\"\n)\n\nprint(sample.head())\nprint(sample.shape)\nprint(sample.dtypes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:34.903485Z","iopub.execute_input":"2026-01-11T13:01:34.903887Z","iopub.status.idle":"2026-01-11T13:01:34.955663Z","shell.execute_reply.started":"2026-01-11T13:01:34.903858Z","shell.execute_reply":"2026-01-11T13:01:34.954472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df = submission_df.copy()\n\n# enforce column order\nfinal_df = final_df[[\"id\", \"value\"]]\n\n# remove NaN / Inf\nfinal_df[\"value\"] = final_df[\"value\"].replace([np.inf, -np.inf], 0.0)\nfinal_df[\"value\"] = final_df[\"value\"].fillna(0.0)\n\n# clip extreme values (safe for ECG)\nfinal_df[\"value\"] = final_df[\"value\"].clip(-5, 5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:39.055727Z","iopub.execute_input":"2026-01-11T13:01:39.056049Z","iopub.status.idle":"2026-01-11T13:01:39.074144Z","shell.execute_reply.started":"2026-01-11T13:01:39.056020Z","shell.execute_reply":"2026-01-11T13:01:39.073114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enforce_length(signal, target_len):\n    if len(signal) > target_len:\n        return signal[:target_len]\n    elif len(signal) < target_len:\n        pad = target_len - len(signal)\n        return np.pad(signal, (0, pad), mode=\"edge\")\n    return signal\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:42.160527Z","iopub.execute_input":"2026-01-11T13:01:42.160948Z","iopub.status.idle":"2026-01-11T13:01:42.166591Z","shell.execute_reply.started":"2026-01-11T13:01:42.160916Z","shell.execute_reply":"2026-01-11T13:01:42.165645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"signal = ensemble_signals(variants)\n\n# vertical alignment only\nsignal = signal - np.mean(signal)\n\n# 🔥 FORCE exact length\nsignal = enforce_length(signal, sig_len)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:45.285109Z","iopub.execute_input":"2026-01-11T13:01:45.285460Z","iopub.status.idle":"2026-01-11T13:01:45.291970Z","shell.execute_reply.started":"2026-01-11T13:01:45.285428Z","shell.execute_reply":"2026-01-11T13:01:45.290691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = []\n\ngrouped = test_meta.groupby(\"id\")\n\nfor ecg_id, rows in tqdm(grouped, desc=\"Inference\"):\n    img = load_image(ecg_id, split=\"test\")\n    lead_imgs = split_leads(img)\n\n    for _, row in rows.iterrows():\n        lead = row.lead\n        fs = row.fs\n        sig_len = row.number_of_rows\n\n        lead_img = lead_imgs[lead]\n\n        variants = reconstruct_variants(lead_img, fs, sig_len)\n        signal = ensemble_signals(variants)\n\n        # vertical offset removal\n        signal = signal - np.mean(signal)\n\n        # 🔥 CRITICAL FIX\n        signal = enforce_length(signal, sig_len)\n\n        for i in range(sig_len):\n            submission.append({\n                \"id\": f\"{ecg_id}_{i}_{lead}\",\n                \"value\": float(signal[i])\n            })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:52.352362Z","iopub.execute_input":"2026-01-11T13:01:52.352750Z","iopub.status.idle":"2026-01-11T13:01:53.649282Z","shell.execute_reply.started":"2026-01-11T13:01:52.352711Z","shell.execute_reply":"2026-01-11T13:01:53.648427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame(submission)\n\n# duplicates\nprint(\"Duplicate IDs:\", submission_df[\"id\"].duplicated().sum())\n\n# missing / extra\nexpected_ids = set(\n    f\"{row.id}_{i}_{row.lead}\"\n    for _, row in test_meta.iterrows()\n    for i in range(row.number_of_rows)\n)\nsubmitted_ids = set(submission_df[\"id\"])\n\nprint(\"Missing:\", len(expected_ids - submitted_ids))\nprint(\"Extra:\", len(submitted_ids - expected_ids))\n\n# NaN / Inf\nprint(\"NaNs:\", submission_df[\"value\"].isna().sum())\nprint(\"Infs:\", np.isinf(submission_df[\"value\"]).sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:01:59.104578Z","iopub.execute_input":"2026-01-11T13:01:59.105576Z","iopub.status.idle":"2026-01-11T13:01:59.881494Z","shell.execute_reply.started":"2026-01-11T13:01:59.105533Z","shell.execute_reply":"2026-01-11T13:01:59.880542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_parquet(\"submission.parquet\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:03.363592Z","iopub.execute_input":"2026-01-11T13:02:03.364805Z","iopub.status.idle":"2026-01-11T13:02:03.411015Z","shell.execute_reply.started":"2026-01-11T13:02:03.364735Z","shell.execute_reply":"2026-01-11T13:02:03.410043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expected_rows = test_meta[\"number_of_rows\"].sum()\nactual_rows = len(submission_df)\n\nprint(\"Expected:\", expected_rows)\nprint(\"Actual  :\", actual_rows)\n\nassert actual_rows == expected_rows\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:06.670471Z","iopub.execute_input":"2026-01-11T13:02:06.671291Z","iopub.status.idle":"2026-01-11T13:02:06.677134Z","shell.execute_reply.started":"2026-01-11T13:02:06.671232Z","shell.execute_reply":"2026-01-11T13:02:06.676117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expected_ids = set(\n    f\"{row.id}_{i}_{row.lead}\"\n    for _, row in test_meta.iterrows()\n    for i in range(row.number_of_rows)\n)\n\nsubmitted_ids = set(submission_df[\"id\"])\n\nprint(\"Missing:\", len(expected_ids - submitted_ids))\nprint(\"Extra  :\", len(submitted_ids - expected_ids))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:10.039788Z","iopub.execute_input":"2026-01-11T13:02:10.040116Z","iopub.status.idle":"2026-01-11T13:02:10.754811Z","shell.execute_reply.started":"2026-01-11T13:02:10.040087Z","shell.execute_reply":"2026-01-11T13:02:10.753778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_parquet(\"submission.parquet\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:13.931943Z","iopub.execute_input":"2026-01-11T13:02:13.932898Z","iopub.status.idle":"2026-01-11T13:02:13.986561Z","shell.execute_reply.started":"2026-01-11T13:02:13.932858Z","shell.execute_reply":"2026-01-11T13:02:13.985531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfinal_df = submission_df.copy()\n\n# EXACT column order\nfinal_df = final_df[[\"id\", \"value\"]]\n\n# FORCE correct dtypes\nfinal_df[\"id\"] = final_df[\"id\"].astype(str)\nfinal_df[\"value\"] = final_df[\"value\"].astype(np.float32)\n\n# FINAL sanity\nprint(final_df.dtypes)\nprint(final_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:17.669135Z","iopub.execute_input":"2026-01-11T13:02:17.669920Z","iopub.status.idle":"2026-01-11T13:02:17.690910Z","shell.execute_reply.started":"2026-01-11T13:02:17.669878Z","shell.execute_reply":"2026-01-11T13:02:17.689740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = pd.read_parquet(\n    \"/kaggle/input/physionet-ecg-image-digitization/sample_submission.parquet\"\n)\n\nprint(sample.dtypes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:21.036951Z","iopub.execute_input":"2026-01-11T13:02:21.037310Z","iopub.status.idle":"2026-01-11T13:02:21.086321Z","shell.execute_reply.started":"2026-01-11T13:02:21.037279Z","shell.execute_reply":"2026-01-11T13:02:21.085170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.to_parquet(\n    \"submission.parquet\",\n    engine=\"pyarrow\",\n    compression=\"snappy\",\n    index=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:25.762028Z","iopub.execute_input":"2026-01-11T13:02:25.762350Z","iopub.status.idle":"2026-01-11T13:02:25.804717Z","shell.execute_reply.started":"2026-01-11T13:02:25.762321Z","shell.execute_reply":"2026-01-11T13:02:25.803643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# reload and verify\ncheck = pd.read_parquet(\"submission.parquet\")\n\nassert list(check.columns) == [\"id\", \"value\"]\nassert check[\"id\"].dtype == object\nassert check[\"value\"].dtype == np.float32\nassert len(check) == test_meta[\"number_of_rows\"].sum()\n\nprint(\"PARQUET VERIFIED — READY TO UPLOAD\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:28.939397Z","iopub.execute_input":"2026-01-11T13:02:28.939773Z","iopub.status.idle":"2026-01-11T13:02:28.985061Z","shell.execute_reply.started":"2026-01-11T13:02:28.939740Z","shell.execute_reply":"2026-01-11T13:02:28.984164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# rebuild final dataframe\nfinal_df = submission_df.copy()\n\n# exact column order\nfinal_df = final_df[[\"id\", \"value\"]]\n\n# enforce correct dtypes\nfinal_df[\"id\"] = final_df[\"id\"].astype(str)\nfinal_df[\"value\"] = final_df[\"value\"].astype(float)   # <-- FLOAT64\n\n# sanity check\nprint(final_df.dtypes)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:32.361523Z","iopub.execute_input":"2026-01-11T13:02:32.361912Z","iopub.status.idle":"2026-01-11T13:02:32.381658Z","shell.execute_reply.started":"2026-01-11T13:02:32.361880Z","shell.execute_reply":"2026-01-11T13:02:32.380521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.to_parquet(\"submission.parquet\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:36.062531Z","iopub.execute_input":"2026-01-11T13:02:36.063393Z","iopub.status.idle":"2026-01-11T13:02:36.107703Z","shell.execute_reply.started":"2026-01-11T13:02:36.063358Z","shell.execute_reply":"2026-01-11T13:02:36.106605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"check = pd.read_parquet(\"submission.parquet\")\n\nprint(check.dtypes)\nprint(len(check), test_meta[\"number_of_rows\"].sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:38.557768Z","iopub.execute_input":"2026-01-11T13:02:38.558119Z","iopub.status.idle":"2026-01-11T13:02:38.608540Z","shell.execute_reply.started":"2026-01-11T13:02:38.558088Z","shell.execute_reply":"2026-01-11T13:02:38.607704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.to_parquet(\"submission.parquet\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T13:02:42.648553Z","iopub.execute_input":"2026-01-11T13:02:42.648955Z","iopub.status.idle":"2026-01-11T13:02:42.694766Z","shell.execute_reply.started":"2026-01-11T13:02:42.648923Z","shell.execute_reply":"2026-01-11T13:02:42.693656Z"}},"outputs":[],"execution_count":null}]}