{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"},{"sourceId":13746387,"sourceType":"datasetVersion","datasetId":8747012},{"sourceId":13816899,"sourceType":"datasetVersion","datasetId":8620533},{"sourceId":271051632,"sourceType":"kernelVersion"},{"sourceId":677607,"sourceType":"modelInstanceVersion","modelInstanceId":513841,"modelId":528480}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Previous verison: [https://www.kaggle.com/code/wasupandceacar/physio-v2-2-public](https://www.kaggle.com/code/wasupandceacar/physio-v2-2-public)\n\nMain Changes:\n\n1. Apply Hint 1 in [https://www.kaggle.com/competitions/physionet-ecg-image-digitization/discussion/624054](https://www.kaggle.com/competitions/physionet-ecg-image-digitization/discussion/624054)\n2. Coresponding changes with resolution change in stage2: network, preprocess, postprocess...","metadata":{}},{"cell_type":"code","source":"!pip uninstall -y tensorflow\n!uv pip install --no-deps --system --no-index --find-links='/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/setup' 'connected-components-3d'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:26:53.417097Z","iopub.execute_input":"2025-12-18T20:26:53.417349Z","iopub.status.idle":"2025-12-18T20:26:55.111329Z","shell.execute_reply.started":"2025-12-18T20:26:53.417333Z","shell.execute_reply":"2025-12-18T20:26:55.110525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile constant.py\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"pydantic\")\n\n# ------------------------\n# Reproducibility\n# ------------------------\nimport kagglehub\nseed = 0\nCUDA0 = \"cuda:0\"\n\ndeterministic = kagglehub.package_import(\n    'wasupandceacar/deterministic'\n).deterministic\n\ndeterministic.init_all(\n    seed,\n    disable_list=['cuda_block']\n)\n\n# Extra safety for torch\nimport torch\ntorch.backends.cudnn.deterministic = True\ntorch.backends.cudnn.benchmark = False\n\n# ------------------------\n# Path & imports\n# ------------------------\nimport sys\nsys.path.append(\n    '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet'\n)\n\nimport os\nimport traceback\nfrom pathlib import Path\nfrom shutil import copyfile\n\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n\n# ------------------------\n# Kaggle mode\n# ------------------------\nif_submit = os.getenv('KAGGLE_IS_COMPETITION_RERUN')\n\nif if_submit:\n    test_meta = Path(\n        \"/kaggle/input/physionet-ecg-image-digitization/test.csv\"\n    )\n    test_dir = Path(\n        \"/kaggle/input/physionet-ecg-image-digitization/test\"\n    )\nelse:\n    test_meta = Path(\n        \"/kaggle/input/physio-test-fake-dataset/test_fake/test.csv\"\n    )\n    test_dir = Path(\n        \"/kaggle/input/physio-test-fake-dataset/test_fake\"\n    )\n\n# ------------------------\n# Load metadata\n# ------------------------\nvalid_df = pd.read_csv(test_meta)\nvalid_df['id'] = valid_df['id'].astype(str)\nvalid_id = valid_df['id'].unique().tolist()\n\n# ------------------------\n# Global config\n# ------------------------\nFLOAT_TYPE = torch.float32\n\nglobal_dict = {\n    \"stage0_dir\": \"/kaggle/working/stage0\",\n    \"stage1_dir\": \"/kaggle/working/stage1\",\n    \"stage2_dir\": \"/kaggle/working/stage2\",\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:26:55.113111Z","iopub.execute_input":"2025-12-18T20:26:55.113368Z","iopub.status.idle":"2025-12-18T20:26:55.119219Z","shell.execute_reply.started":"2025-12-18T20:26:55.113343Z","shell.execute_reply":"2025-12-18T20:26:55.118484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage0.py\n\nfrom constant import *\n\nfrom stage0_model import Net as Stage0Net\nfrom stage0_common import *\n\n# ------------------------\n# Prepare output dir\n# ------------------------\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage0_dir.mkdir(exist_ok=True, parents=True)\n\n# ------------------------\n# Load model\n# ------------------------\nstage0_net = Stage0Net(pretrained=False)\nstage0_net = load_net(\n    stage0_net,\n    '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth'\n)\nstage0_net.to(CUDA0)\nstage0_net.eval()   # IMPORTANT\n\n# ------------------------\n# Inference\n# ------------------------\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = test_dir / f'{sample_id}.png'\n    output_path = stage0_dir / f'{sample_id}.png'\n\n    image = cv2.imread(path, cv2.IMREAD_COLOR_RGB)\n    if image is None:\n        continue\n\n    batch = image_to_batch(image)\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage0_net(batch)\n\n        rotated, keypoint = output_to_predict(image, batch, output)\n        normalised, _, _ = normalise_by_homography(rotated, keypoint)\n\n        cv2.imwrite(\n            output_path,\n            cv2.cvtColor(normalised, cv2.COLOR_RGB2BGR)\n        )\n\n    except Exception:\n        traceback.print_exc()\n        copyfile(path, output_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:26:55.119911Z","iopub.execute_input":"2025-12-18T20:26:55.120163Z","iopub.status.idle":"2025-12-18T20:26:55.136075Z","shell.execute_reply.started":"2025-12-18T20:26:55.12014Z","shell.execute_reply":"2025-12-18T20:26:55.135431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage1.py\n\nfrom constant import *\n\nfrom stage1_model import Net as Stage1Net\nfrom stage1_common import *\n\n# ------------------------\n# Prepare output dir\n# ------------------------\nstage0_dir = Path(global_dict[\"stage0_dir\"])\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage1_dir.mkdir(exist_ok=True, parents=True)\n\n# ------------------------\n# Load model\n# ------------------------\nstage1_net = Stage1Net(pretrained=False)\nstage1_net = load_net(\n    stage1_net,\n    '/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth'\n)\nstage1_net.to(CUDA0)\nstage1_net.eval()   # IMPORTANT\n\n# ------------------------\n# Inference\n# ------------------------\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage0_dir / f'{sample_id}.png'\n    output_path = stage1_dir / f'{sample_id}.png'\n\n    image = cv2.imread(path, cv2.IMREAD_COLOR_RGB)\n    if image is None:\n        continue\n\n    batch = {\n        'image': torch.from_numpy(\n            np.ascontiguousarray(image.transpose(2, 0, 1))\n        ).unsqueeze(0).to(CUDA0)\n    }\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage1_net(batch)\n\n        gridpoint_xy, _ = output_to_predict(image, batch, output)\n        rectified = rectify_image(image, gridpoint_xy)\n\n        cv2.imwrite(\n            output_path,\n            cv2.cvtColor(rectified, cv2.COLOR_RGB2BGR)\n        )\n\n    except Exception:\n        traceback.print_exc()\n        copyfile(path, output_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:26:55.137509Z","iopub.execute_input":"2025-12-18T20:26:55.137794Z","iopub.status.idle":"2025-12-18T20:26:55.154555Z","shell.execute_reply.started":"2025-12-18T20:26:55.137778Z","shell.execute_reply":"2025-12-18T20:26:55.153794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile stage2.py\n\nfrom constant import *\n\nimport torchvision.transforms as T\nfrom scipy.signal import savgol_filter\n\nfrom stage2_model import *\nfrom stage2_common import *\n\n# -------------------------------------------------\n# Dynamic zero-line estimation\n# -------------------------------------------------\ndef estimate_zero_line(pixel_map):\n    \"\"\"\n    pixel_map: (H, W) probability map for one lead\n    returns estimated baseline y-coordinate\n    \"\"\"\n    ys = np.where(pixel_map < 0.1)[0]\n    if len(ys) == 0:\n        return None\n    return np.median(ys)\n\n# -------------------------------------------------\n# Model\n# -------------------------------------------------\nclass Net3(nn.Module):\n    def __init__(self, pretrained=True):\n        super(Net3, self).__init__()\n        encoder_dim = [64, 128, 256, 512]\n        decoder_dim = [128, 64, 32, 16]\n\n        self.encoder = timm.create_model(\n            model_name='resnet34.a3_in1k',\n            pretrained=pretrained,\n            in_chans=3,\n            num_classes=0,\n            global_pool=''\n        )\n\n        self.decoder = MyCoordUnetDecoder(\n            in_channel=encoder_dim[-1],\n            skip_channel=encoder_dim[:-1][::-1] + [0],\n            out_channel=decoder_dim,\n            scale=[2, 2, 2, 2]\n        )\n\n        self.pixel = nn.Conv2d(decoder_dim[-1], 4, 1)\n\n    def forward(self, image):\n        encode = encode_with_resnet(self.encoder, image)\n        last, _ = self.decoder(\n            feature=encode[-1],\n            skip=encode[:-1][::-1] + [None]\n        )\n        pixel = self.pixel(last)\n        return pixel\n\n# -------------------------------------------------\n# Paths\n# -------------------------------------------------\nstage1_dir = Path(global_dict[\"stage1_dir\"])\nstage2_dir = Path(global_dict[\"stage2_dir\"])\nstage2_dir.mkdir(exist_ok=True, parents=True)\n\n# -------------------------------------------------\n# Load model\n# -------------------------------------------------\nstage2_net = Net3(pretrained=False).to(CUDA0)\nmodel_path = \"/kaggle/input/physio-seg-public/pytorch/net3_009_4200/1/iter_0004200.pt\"\nstage2_net.load_state_dict(torch.load(model_path))\nstage2_net.eval()\n\n# -------------------------------------------------\n# Constants\n# -------------------------------------------------\nx0, x1 = 0, 2176\ny0, y1 = 0, 1696\n\nzero_mv = [703.5, 987.5, 1271.5, 1531.5]   # fallback\nmv_to_pixel = 78.5\n\nt0, t1 = 235, 4161\n\nresize = T.Resize(\n    (1696, 4352),\n    interpolation=T.InterpolationMode.BILINEAR\n)\n\n# -------------------------------------------------\n# Inference\n# -------------------------------------------------\nfor n, sample_id in enumerate(tqdm(valid_id)):\n    path = stage1_dir / f'{sample_id}.png'\n    output_path = stage2_dir / f'{sample_id}.npy'\n\n    image = cv2.imread(path, cv2.IMREAD_COLOR_RGB)\n    if image is None:\n        continue\n\n    length = valid_df[\n        (valid_df['id'] == sample_id) &\n        (valid_df['lead'] == 'II')\n    ].iloc[0].number_of_rows\n\n    image = image[y0:y1, x0:x1] / 255.0\n\n    batch = resize(\n        torch.from_numpy(\n            np.ascontiguousarray(image.transpose(2, 0, 1))\n        ).unsqueeze(0)\n    ).float().to(CUDA0)\n\n    try:\n        with torch.no_grad(), torch.amp.autocast('cuda', dtype=FLOAT_TYPE):\n            output = stage2_net(batch)\n\n        pixel = torch.sigmoid(output).float().cpu().numpy()[0]\n\n        # -------- dynamic zero-line --------\n        dynamic_zero = []\n        for ch in range(4):\n            zl = estimate_zero_line(pixel[ch])\n            if zl is None:\n                zl = zero_mv[ch]\n            dynamic_zero.append(zl)\n\n        dynamic_zero = np.array(dynamic_zero)\n\n        # -------- pixel → signal --------\n        series_in_pixel = pixel_to_series(\n            pixel[..., t0:t1],\n            dynamic_zero,\n            length\n        )\n\n        series = (\n            dynamic_zero.reshape(4, 1) - series_in_pixel\n        ) / mv_to_pixel\n\n        # -------- smoothing --------\n        series = savgol_filter(\n            series,\n            window_length=11,\n            polyorder=3,\n            axis=1\n        )\n\n        np.save(output_path, series)\n\n    except Exception:\n        traceback.print_exc()\n        series = np.zeros((4, length))\n        np.save(output_path, series)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:26:55.155521Z","iopub.execute_input":"2025-12-18T20:26:55.155743Z","iopub.status.idle":"2025-12-18T20:27:05.419733Z","shell.execute_reply.started":"2025-12-18T20:26:55.155729Z","shell.execute_reply":"2025-12-18T20:27:05.418876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python stage0.py\n!python stage1.py\n!python stage2.py","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:27:05.42055Z","iopub.execute_input":"2025-12-18T20:27:05.420818Z","iopub.status.idle":"2025-12-18T20:28:06.29049Z","shell.execute_reply.started":"2025-12-18T20:27:05.420795Z","shell.execute_reply":"2025-12-18T20:28:06.289588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\n# ==============================\n# 1️⃣ عرض التنبؤات مقابل GT\n# ==============================\ndef show_pred_gt(pred, gt):\n    \"\"\"\n    عرض التنبؤات مقابل القيم الحقيقية لكل Leads الـ 12\n    \"\"\"\n    n_leads = pred.shape[1]\n    fig = make_subplots(\n        rows=n_leads, cols=1, \n        subplot_titles=[f'Lead {i+1}' for i in range(n_leads)]\n    )\n\n    for i in range(n_leads):\n        fig.add_trace(\n            go.Scatter(y=pred[:, i], mode='lines', name=f'Pred Lead {i+1}', line=dict(color='blue')),\n            row=i+1, col=1\n        )\n        fig.add_trace(\n            go.Scatter(y=gt[:, i], mode='lines', name=f'GT Lead {i+1}', line=dict(color='red')),\n            row=i+1, col=1\n        )\n    \n    fig.update_layout(\n        height=1200, \n        showlegend=False,\n        title=\"Predictions vs Ground Truth\",\n    )\n    fig.show(renderer='iframe')\n\n# ==============================\n# 2️⃣ تحويل 4 قنوات لـ 12 قناة\n# ==============================\ndef expand_4_to_12(pred4):\n    \"\"\"\n    تحويل المخرجات من 4 قنوات ECG إلى 12 قناة\n    بشكل مرن لأي عدد عينات\n    \"\"\"\n    n_samples = pred4.shape[0]\n    pred12 = np.zeros((n_samples, 12))\n    step = n_samples // 4\n\n    # توزيع القنوات\n    pred12[:step, [0, 1, 2, 3]] = pred4[:step, [0, 3, 2, 1]]\n    pred12[step:2*step, [4, 5, 6, 7]] = pred4[step:2*step, [0, 1, 2, 3]]\n    pred12[2*step:3*step, [8, 9, 10, 11]] = pred4[2*step:3*step, [0, 1, 2, 3]]\n\n    # التعامل مع أي عينات متبقية\n    if 3*step < n_samples:\n        remaining = n_samples - 3*step\n        pred12[3*step:, :remaining] = pred4[3*step:, :remaining]\n\n    return pred12\n\n# ==============================\n# 3️⃣ إنشاء dict للسلسلة لكل lead\n# ==============================\ndef series_dict(series):\n    \"\"\"\n    تقسيم كل سلسلة إلى الـ Leads المناسبة\n    \"\"\"\n    lead_map = {\n        0: ['I', 'aVR', 'V1', 'V4'],\n        1: ['II', 'aVL', 'V2', 'V5'],\n        2: ['III', 'aVF', 'V3', 'V6']\n    }\n    \n    series_by_lead = {}\n    for l_idx, leads in lead_map.items():\n        splits = np.array_split(series[l_idx], len(leads))\n        for lead_name, s in zip(leads, splits):\n            series_by_lead[lead_name] = s\n\n    # إضافة II من القناة الرابعة لو موجودة\n    if len(series) > 3:\n        series_by_lead['II'] = series[3]\n    \n    return series_by_lead\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:28:06.291559Z","iopub.execute_input":"2025-12-18T20:28:06.29185Z","iopub.status.idle":"2025-12-18T20:28:06.302907Z","shell.execute_reply.started":"2025-12-18T20:28:06.291806Z","shell.execute_reply":"2025-12-18T20:28:06.302114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from constant import *\n\nstage2_dir = Path(global_dict[\"stage2_dir\"])\n\nsubmit_df = list()\ngb = valid_df.groupby('id')\n\nshow = True\n\nfor rec_idx, (sample_id, df) in enumerate(tqdm(gb)):\n    series = np.load(stage2_dir / f'{sample_id}.npy')\n\n    if not if_submit and show:\n        pred = np.transpose(series, axes=(1, 0))\n        pred = expand_4_to_12(pred)\n        gt = pd.read_csv(\"/kaggle/input/physio-test-fake-dataset/7663343_inp1.csv\").fillna(0).values\n        show_pred_gt(pred, gt)\n        show = False\n    \n    series_by_lead = series_dict(series)\n\n    for _, d in df.iterrows():\n        s = series_by_lead[d.lead]\n        if len(s) != d.number_of_rows:\n            x_old = np.linspace(0.0, 1.0, len(s))\n            x_new = np.linspace(0.0, 1.0, d.number_of_rows)\n            s = np.interp(x_new, x_old, s)\n        row_id = [f'{sample_id}_{t}_{d.lead}' for t in range(d.number_of_rows)]\n        this_df = pd.DataFrame({\n            'id': row_id,\n            'value': s,\n        })\n        submit_df.append(this_df)\n\nsubmit_df = pd.concat(submit_df, axis=0, ignore_index=True, sort=False, copy=False)\nsubmit_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:28:06.303593Z","iopub.execute_input":"2025-12-18T20:28:06.303828Z","iopub.status.idle":"2025-12-18T20:28:08.594185Z","shell.execute_reply.started":"2025-12-18T20:28:06.303813Z","shell.execute_reply":"2025-12-18T20:28:08.593299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# تحميل الـ submission النهائي\nsub = pd.read_csv('submission.csv')\n\n# ==========================\n# 1️⃣ التحقق من NaN أو قيم غير صالحة\n# ==========================\nnan_count = sub['value'].isna().sum()\nprint(f\"عدد القيم NaN في الـ submission: {nan_count}\")\n\n# ==========================\n# 2️⃣ التأكد من وجود كل lead لكل sample\n# ==========================\n# استخراج أسماء الـ leads من الـ IDs\nsub['lead'] = sub['id'].apply(lambda x: x.split('_')[-1])\nsub['sample_id'] = sub['id'].apply(lambda x: '_'.join(x.split('_')[:-2]))\n\nexpected_leads = ['I','II','III','aVR','aVL','aVF','V1','V2','V3','V4','V5','V6']\n\n# تحقق لكل sample\nmissing_leads = {}\nfor sample_id, group in sub.groupby('sample_id'):\n    present_leads = set(group['lead'].unique())\n    missing = set(expected_leads) - present_leads\n    if missing:\n        missing_leads[sample_id] = missing\n\nif missing_leads:\n    print(\"يوجد samples مفقود لها leads:\")\n    for k,v in missing_leads.items():\n        print(f\"{k}: missing {v}\")\nelse:\n    print(\"كل الـ samples تحتوي على الـ 12 leads ✅\")\n\n# ==========================\n# 3️⃣ التحقق من طول كل lead\n# ==========================\nlength_issues = {}\nfor sample_id, group in sub.groupby('sample_id'):\n    for lead in expected_leads:\n        n_rows_expected = group[group['lead']==lead].shape[0]\n        if n_rows_expected != len(group[group['lead']==lead]):\n            length_issues[f\"{sample_id}_{lead}\"] = n_rows_expected\n\nif length_issues:\n    print(\"يوجد مشاكل في طول بعض الـ leads:\")\n    print(length_issues)\nelse:\n    print(\"طول كل lead صحيح ✅\")\n\n# ==========================\n# 4️⃣ نظرة سريعة على البيانات\n# ==========================\nprint(\"\\nأول 20 صف في الـ submission:\")\nprint(sub.head(20))\nprint(f\"\\nإجمالي عدد الصفوف: {len(sub)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T20:28:08.595323Z","iopub.execute_input":"2025-12-18T20:28:08.595668Z","iopub.status.idle":"2025-12-18T20:28:08.844153Z","shell.execute_reply.started":"2025-12-18T20:28:08.595636Z","shell.execute_reply":"2025-12-18T20:28:08.843406Z"}},"outputs":[],"execution_count":null}]}