{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl, numpy as np, matplotlib.pyplot as plt\nfrom pathlib import Path\nimport matplotlib.image as mpimg\nimport random\n\n!ls /kaggle/input/physionet-ecg-image-digitization","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:29:45.466842Z","iopub.execute_input":"2025-10-27T08:29:45.467766Z","iopub.status.idle":"2025-10-27T08:29:45.601448Z","shell.execute_reply.started":"2025-10-27T08:29:45.467726Z","shell.execute_reply":"2025-10-27T08:29:45.600376Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"ROOT = Path('/kaggle/input/physionet-ecg-image-digitization')\nSEED = 10304\nrandom.seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:29:45.603216Z","iopub.execute_input":"2025-10-27T08:29:45.603532Z","iopub.status.idle":"2025-10-27T08:29:45.608783Z","shell.execute_reply.started":"2025-10-27T08:29:45.603507Z","shell.execute_reply":"2025-10-27T08:29:45.607817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pl.read_csv(ROOT / 'train.csv')\ntest_df = pl.read_csv(ROOT / 'test.csv')\nrandom_train_id = random.choice(train_df['id'])\nrandom_test_id = random.choice(test_df['id'])\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:29:45.609726Z","iopub.execute_input":"2025-10-27T08:29:45.610076Z","iopub.status.idle":"2025-10-27T08:29:45.634699Z","shell.execute_reply.started":"2025-10-27T08:29:45.610055Z","shell.execute_reply":"2025-10-27T08:29:45.633722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:29:45.636728Z","iopub.execute_input":"2025-10-27T08:29:45.636991Z","iopub.status.idle":"2025-10-27T08:29:45.643674Z","shell.execute_reply.started":"2025-10-27T08:29:45.636970Z","shell.execute_reply":"2025-10-27T08:29:45.642833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sampling frequency distribution","metadata":{}},{"cell_type":"code","source":"plt.hist(\n    train_df['fs'].to_numpy(),\n    bins=train_df['fs'].unique().count(),\n    edgecolor='black'\n)\nplt.title(\"Histogram of `fs` column\")\nplt.xlabel(\"fs\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:29:45.644461Z","iopub.execute_input":"2025-10-27T08:29:45.644693Z","iopub.status.idle":"2025-10-27T08:29:45.839337Z","shell.execute_reply.started":"2025-10-27T08:29:45.644667Z","shell.execute_reply":"2025-10-27T08:29:45.838548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Show a random image","metadata":{}},{"cell_type":"markdown","source":"### Train","metadata":{}},{"cell_type":"code","source":"n = {\n 1: 'Original color ECG image generated by ECG-image-kit',\n 3: 'Image printed in color and scanned in color',\n 4: 'Image printed in color and scanned in black and white',\n 5: 'Mobile photo of color printed image',\n 6: 'Mobile photo of ECG on the screen of laptop',\n 9: 'Mobile photo of stained and soaked printed ECG',\n 10: 'Mobile photo of printed ECG with extensive damage',\n 11: 'Scan of printed ECG image with mold in color',\n 12: 'Scan of printed ECG image with mold in black and white'\n}\n\n\nfig, axes = plt.subplots(3,3, figsize=(12,8))\nfig.suptitle(f\"ECG Variations for ID {random_train_id}\", fontsize=14, weight='bold')\n\nfor i, ax in zip(n.keys(), axes.ravel()):\n    img_path = ROOT / 'train' / str(random_train_id) / f'{random_train_id}-{i:04d}.png'\n    ax.imshow(mpimg.imread(img_path))\n    ax.set_title(n[i], fontsize=9)\n    ax.axis('off')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:29:45.840487Z","iopub.execute_input":"2025-10-27T08:29:45.840819Z","iopub.status.idle":"2025-10-27T08:30:07.150872Z","shell.execute_reply.started":"2025-10-27T08:29:45.840792Z","shell.execute_reply":"2025-10-27T08:30:07.149771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test","metadata":{}},{"cell_type":"code","source":"plt.imshow(\n    mpimg.imread(ROOT / 'test' / f'{random_test_id}.png')\n)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:30:07.152007Z","iopub.execute_input":"2025-10-27T08:30:07.152319Z","iopub.status.idle":"2025-10-27T08:30:08.063033Z","shell.execute_reply.started":"2025-10-27T08:30:07.152267Z","shell.execute_reply":"2025-10-27T08:30:08.062012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Training dataframe","metadata":{}},{"cell_type":"code","source":"fs = train_df.filter(pl.col(\"id\") == random_train_id).row(0)[1]          # sampling frequency\nsig_len = train_df.filter(pl.col(\"id\") == random_train_id).row(0)[2]     # total number of samples\n\nrandom_train_id_df = pl.read_csv(\n    ROOT / \"train\" / str(random_train_id) / f\"{random_train_id}.csv\"\n).with_columns(pl.Series(\"time\", np.arange(sig_len) / fs))\n\nrandom_train_id_df = random_train_id_df.select([\"time\"] + [c for c in random_train_id_df.columns if c != \"time\"])\nrandom_train_id_df = random_train_id_df.select([\n    pl.when(pl.col(c).is_not_null())\n      .then(pl.col(c).cast(pl.Float64, strict=False))\n      .otherwise(None)\n      .alias(c)\n    for c in random_train_id_df.columns\n])\n\nrandom_train_id_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:30:08.064078Z","iopub.execute_input":"2025-10-27T08:30:08.064393Z","iopub.status.idle":"2025-10-27T08:30:08.105899Z","shell.execute_reply.started":"2025-10-27T08:30:08.064370Z","shell.execute_reply":"2025-10-27T08:30:08.104918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Null range","metadata":{}},{"cell_type":"code","source":"def get_non_null_ranges(df: pl.DataFrame):\n    ranges = {}\n    n = df.height\n    \n    for col in df.columns:\n        mask = df[col].is_not_null().to_numpy()\n        if mask.any():\n            start = mask.argmax()  # first True index\n            end = n - mask[::-1].argmax()  # last True index (exclusive)\n            ranges[col] = (start, end, end - start)\n        else:\n            ranges[col] = None\n    return ranges\n\n\nranges = get_non_null_ranges(random_train_id_df)\nfor col, (start, end, length) in ranges.items():\n    print(f\"{col:>4s}: non-null from {start} to {end-1} (length={length})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:30:08.106921Z","iopub.execute_input":"2025-10-27T08:30:08.107224Z","iopub.status.idle":"2025-10-27T08:30:08.116826Z","shell.execute_reply.started":"2025-10-27T08:30:08.107205Z","shell.execute_reply":"2025-10-27T08:30:08.115873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Null count","metadata":{}},{"cell_type":"code","source":"random_train_id_df.select([\n    pl.col(c).null_count().alias(c) for c in random_train_id_df.columns\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:30:08.119269Z","iopub.execute_input":"2025-10-27T08:30:08.119540Z","iopub.status.idle":"2025-10-27T08:30:08.139704Z","shell.execute_reply.started":"2025-10-27T08:30:08.119520Z","shell.execute_reply":"2025-10-27T08:30:08.138755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot (train)","metadata":{}},{"cell_type":"code","source":"def extract_non_null_df(df: pl.DataFrame, col_name: str) -> pl.DataFrame:\n    mask = df[col_name].is_not_null().to_numpy()\n    if not mask.any():\n        return None\n    start = mask.argmax()\n    end = len(mask) - mask[::-1].argmax()\n    return df.slice(start, end - start).select([\"time\", col_name])\n\nsplit_dfs = {}\n\nfor col in [c for c in random_train_id_df.columns if c != \"time\"]:\n    sub_df = extract_non_null_df(random_train_id_df, col)\n    if sub_df is not None:\n        split_dfs[col] = sub_df\n\nfig, axes = plt.subplots(6, 2, figsize=(14, 12))\naxes = axes.flatten()  # flatten grid into a list of 12 axes\n\nfor i, (lead, df_lead) in enumerate(split_dfs.items()):\n    # Convert to numpy arrays for plotting\n    time_vals = df_lead[\"time\"].to_numpy()\n    signal_vals = df_lead[lead].to_numpy()\n\n    # Plot each lead\n    ax = axes[i]\n    ax.plot(time_vals, signal_vals, linewidth=0.8)\n    ax.set_title(lead)\n    ax.set_xlabel(\"Time (s)\")\n    ax.set_ylabel(\"mV\")\n    ax.grid(True)\n\n# Hide any unused subplots (in case fewer than 12)\nfor j in range(i + 1, len(axes)):\n    axes[j].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T08:30:08.140584Z","iopub.execute_input":"2025-10-27T08:30:08.140910Z","iopub.status.idle":"2025-10-27T08:30:09.877722Z","shell.execute_reply.started":"2025-10-27T08:30:08.140882Z","shell.execute_reply":"2025-10-27T08:30:09.876633Z"}},"outputs":[],"execution_count":null}]}