{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"},{"sourceId":13746387,"sourceType":"datasetVersion","datasetId":8747012},{"sourceId":13816899,"sourceType":"datasetVersion","datasetId":8620533},{"sourceId":271051632,"sourceType":"kernelVersion"}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Base on [https://www.kaggle.com/code/seshurajup/henkgck-submission-v4-credits-to-hengck](https://www.kaggle.com/code/seshurajup/henkgck-submission-v4-credits-to-hengck)\n\nMain Changes:\n\n1. Seperate 3 stages to 3 files to avoid multiple models in GPU\n2. Fix global seed to 0\n<!-- 3. Set `FLOAT_TYPE` from `float32` to `bfloat16` -->\n4. Set `mv_to_pixel = 78.3` and `t0, t1 = 117, 2081`\n5. Remove `filter_series_by_limits` cause I find it don't help LB\n6. Many code refactoring","metadata":{}},{"cell_type":"code","source":"!pip uninstall -y tensorflow\n!uv pip install --no-deps --system --no-index --find-links='/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/setup' 'connected-components-3d'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:42:24.371376Z","iopub.execute_input":"2025-11-30T16:42:24.371577Z","iopub.status.idle":"2025-11-30T16:42:46.382557Z","shell.execute_reply.started":"2025-11-30T16:42:24.371554Z","shell.execute_reply":"2025-11-30T16:42:46.381852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile constant.py\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"pydantic\")\n\nimport kagglehub\nseed = 0\nCUDA0 = \"cuda:0\"\ndeterministic = kagglehub.package_import('wasupandceacar/deterministic').deterministic\ndeterministic.init_all(seed, disable_list=['cuda_block'])\n\nimport sys\nsys.path.append('/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet')\n\nimport os\nimport traceback\nfrom pathlib import Path\nfrom shutil import copyfile\nimport torch\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n\nif_submit = os.getenv('KAGGLE_IS_COMPETITION_RERUN')\n\nif if_submit:\n    test_meta = Path(\"/kaggle/input/physionet-ecg-image-digitization/test.csv\")\n    test_dir = Path(\"/kaggle/input/physionet-ecg-image-digitization/test\")\nelse:\n    test_meta = Path(\"/kaggle/input/physio-test-fake-dataset/test_fake/test.csv\")\n    test_dir = Path(\"/kaggle/input/physio-test-fake-dataset/test_fake\")\n\nvalid_df = pd.read_csv(test_meta)\nvalid_df['id'] = valid_df['id'].astype(str) \nvalid_id = valid_df['id'].unique().tolist()\n\nFLOAT_TYPE = torch.float32\n\nglobal_dict = {\n    \"stage0_dir\": \"/kaggle/working/stage0\",\n    \"stage1_dir\": \"/kaggle/working/stage1\",\n    \"stage2_dir\": \"/kaggle/working/stage2\",\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:42:46.384988Z","iopub.execute_input":"2025-11-30T16:42:46.385627Z","iopub.status.idle":"2025-11-30T16:42:46.391599Z","shell.execute_reply.started":"2025-11-30T16:42:46.385599Z","shell.execute_reply":"2025-11-30T16:42:46.390925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage0.py\n\nfrom constant import *\n\nfrom stage0_model import Net as Stage0Net\nfrom stage0_common import *\n\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage0_dir.mkdir(exist_ok=True)\n\nstage0_net = Stage0Net(pretrained=False)\nstage0_net = load_net(stage0_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth')\nstage0_net.to(CUDA0)\n\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = test_dir / f'{sample_id}.png'\n    output_path = stage0_dir / f'{sample_id}.png'\n    image = cv2.imread(path, cv2.IMREAD_COLOR_RGB)\n    batch = image_to_batch(image)\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage0_net(batch)\n        rotated, keypoint = output_to_predict(image, batch, output)\n        normalised, _, _ = normalise_by_homography(rotated, keypoint)\n        cv2.imwrite(output_path, cv2.cvtColor(normalised, cv2.COLOR_RGB2BGR))\n    except:\n        traceback.print_exc()\n        copyfile(path, output_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:42:46.392338Z","iopub.execute_input":"2025-11-30T16:42:46.392556Z","iopub.status.idle":"2025-11-30T16:42:46.436849Z","shell.execute_reply.started":"2025-11-30T16:42:46.392531Z","shell.execute_reply":"2025-11-30T16:42:46.436012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage1.py\n\nfrom constant import *\n\nfrom stage1_model import Net as Stage1Net\nfrom stage1_common import *\n\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage1_dir.mkdir(exist_ok=True)\n\nstage1_net = Stage1Net(pretrained=False)\nstage1_net = load_net(stage1_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth')\nstage1_net.to(CUDA0)\n\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage0_dir / f'{sample_id}.png'\n    output_path = stage1_dir / f'{sample_id}.png'\n    image = cv2.imread(path, cv2.IMREAD_COLOR_RGB)\n    batch = {'image': torch.from_numpy(np.ascontiguousarray(image.transpose(2, 0, 1))).unsqueeze(0)}\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage1_net(batch)\n        gridpoint_xy, _ = output_to_predict(image, batch, output)\n        rectified = rectify_image(image, gridpoint_xy)\n        cv2.imwrite(output_path, cv2.cvtColor(rectified, cv2.COLOR_RGB2BGR))\n    except:\n        traceback.print_exc()\n        copyfile(path, output_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:43:57.954849Z","iopub.execute_input":"2025-11-30T16:43:57.955214Z","iopub.status.idle":"2025-11-30T16:43:57.961169Z","shell.execute_reply.started":"2025-11-30T16:43:57.955180Z","shell.execute_reply":"2025-11-30T16:43:57.960493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage2.py\n\nfrom constant import *\n\nfrom stage2_model import Net as Stage2Net\nfrom stage2_common import *\n\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage2_dir = Path(global_dict[\"stage2_dir\"])\nstage2_dir.mkdir(exist_ok=True)\n\nstage2_net = Stage2Net(pretrained=False)\nstage2_net = load_net(stage2_net, '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage2-00005810.checkpoint.pth')\nstage2_net.to(CUDA0)\n\nx0, x1 = 0, 2176\ny0, y1 = 0, 1696\nzero_mv = [703.5, 987.5, 1271.5, 1531.5]\nmv_to_pixel = 78.3\nt0, t1 = 117, 2081\n\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage1_dir / f'{sample_id}.png'\n    output_path = stage2_dir / f'{sample_id}.npy'\n    image = cv2.imread(path, cv2.IMREAD_COLOR_RGB)\n    length = valid_df[(valid_df['id']==sample_id) & (valid_df['lead']=='II')].iloc[0].number_of_rows\n    image = image[y0:y1, x0:x1]\n    batch = {'image': torch.from_numpy(np.ascontiguousarray(image.transpose(2, 0, 1))).unsqueeze(0)}\n    \n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage2_net(batch)\n        pixel = output['pixel'].float().data.cpu().numpy()[0]\n        series_in_pixel = pixel_to_series(pixel[..., t0:t1], zero_mv, length)\n        series = (np.array(zero_mv).reshape(4, 1) - series_in_pixel) / mv_to_pixel\n        np.save(output_path, series)\n    except:\n        traceback.print_exc()\n        series = np.zeros((4, length))\n        np.save(output_path, series)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:49:34.209947Z","iopub.execute_input":"2025-11-30T16:49:34.210702Z","iopub.status.idle":"2025-11-30T16:49:34.216424Z","shell.execute_reply.started":"2025-11-30T16:49:34.210669Z","shell.execute_reply":"2025-11-30T16:49:34.215810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python stage0.py\n!python stage1.py\n!python stage2.py","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:49:36.569036Z","iopub.execute_input":"2025-11-30T16:49:36.569612Z","iopub.status.idle":"2025-11-30T16:49:56.550850Z","shell.execute_reply.started":"2025-11-30T16:49:36.569587Z","shell.execute_reply":"2025-11-30T16:49:56.550150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\ndef show_pred_gt(pred, gt):\n    fig = make_subplots(rows=12, cols=1, subplot_titles=[f'Lead {i+1}' for i in range(12)])\n    for i in range(12):\n        fig.add_trace(go.Scatter(y=pred[:, i], mode='lines', name=f'Pred Lead {i+1}', line=dict(color='blue')), row=i+1, col=1)\n        fig.add_trace(go.Scatter(y=gt[:, i], mode='lines', name=f'GT Lead {i+1}', line=dict(color='red')), row=i+1, col=1)\n    fig.update_layout(height=1200, showlegend=False)\n    fig.show(renderer='iframe')\n\ndef expand_4_to_12(pred4):\n    pred12 = np.zeros((pred4.shape[0], 12))\n    quat = pred4.shape[0] // 4\n\n    pred12[:quat, 0] = pred4[:quat, 0]\n    pred12[:, 1] = pred4[:, 3]\n    pred12[:quat, 2] = pred4[:quat, 2]\n    pred12[quat:2*quat, 3] = pred4[quat:2*quat, 0]\n    pred12[quat:2*quat, 4] = pred4[quat:2*quat, 1]\n    pred12[quat:2*quat, 5] = pred4[quat:2*quat, 2]\n    pred12[2*quat:3*quat, 6] = pred4[2*quat:3*quat, 0]\n    pred12[2*quat:3*quat, 7] = pred4[2*quat:3*quat, 1]\n    pred12[2*quat:3*quat, 8] = pred4[2*quat:3*quat, 2]\n    pred12[3*quat:4*quat, 9] = pred4[3*quat:4*quat, 0]\n    pred12[3*quat:4*quat, 10] = pred4[3*quat:4*quat, 1]\n    pred12[3*quat:4*quat, 11] = pred4[3*quat:4*quat, 2]\n    \n    return pred12\n\ndef series_dict(series):\n    series_by_lead = dict()\n    for l in range(3):\n        lead_names = [\n            ['I',   'aVR', 'V1', 'V4'],\n            ['II',  'aVL', 'V2', 'V5'],\n            ['III', 'aVF', 'V3', 'V6'],\n        ][l]\n        split = np.array_split(series[l], 4)\n        for (k, s) in zip(lead_names, split):\n            series_by_lead[k] = s\n    series_by_lead['II'] = series[3]\n    return series_by_lead","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:50:58.546662Z","iopub.execute_input":"2025-11-30T16:50:58.547073Z","iopub.status.idle":"2025-11-30T16:50:58.670410Z","shell.execute_reply.started":"2025-11-30T16:50:58.547042Z","shell.execute_reply":"2025-11-30T16:50:58.669815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from constant import *\n\nstage2_dir = Path(global_dict[\"stage2_dir\"])\n\nsubmit_df = list()\ngb = valid_df.groupby('id')\n\nshow = True\n\nfor rec_idx, (sample_id, df) in enumerate(tqdm(gb)):\n    series = np.load(stage2_dir / f'{sample_id}.npy')\n\n    if not if_submit and show:\n        pred = np.transpose(series, axes=(1, 0))\n        pred = expand_4_to_12(pred)\n        gt = pd.read_csv(\"/kaggle/input/physio-test-fake-dataset/7663343_inp1.csv\").fillna(0).values\n        show_pred_gt(pred, gt)\n        show = False\n    \n    series_by_lead = series_dict(series)\n\n    for _, d in df.iterrows():\n        s = series_by_lead[d.lead]\n        if len(s) != d.number_of_rows:\n            x_old = np.linspace(0.0, 1.0, len(s))\n            x_new = np.linspace(0.0, 1.0, d.number_of_rows)\n            s = np.interp(x_new, x_old, s)\n        row_id = [f'{sample_id}_{t}_{d.lead}' for t in range(d.number_of_rows)]\n        this_df = pd.DataFrame({\n            'id': row_id,\n            'value': s,\n        })\n        submit_df.append(this_df)\n\nsubmit_df = pd.concat(submit_df, axis=0, ignore_index=True, sort=False, copy=False)\nsubmit_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:51:27.901236Z","iopub.execute_input":"2025-11-30T16:51:27.901487Z","iopub.status.idle":"2025-11-30T16:51:35.412091Z","shell.execute_reply.started":"2025-11-30T16:51:27.901470Z","shell.execute_reply":"2025-11-30T16:51:35.411300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv('submission.csv')\nprint(len(sub))\nsub.head(30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-30T16:51:47.022425Z","iopub.execute_input":"2025-11-30T16:51:47.023062Z","iopub.status.idle":"2025-11-30T16:51:47.305300Z","shell.execute_reply.started":"2025-11-30T16:51:47.023036Z","shell.execute_reply":"2025-11-30T16:51:47.304644Z"}},"outputs":[],"execution_count":null}]}