{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"},{"sourceId":13746387,"sourceType":"datasetVersion","datasetId":8747012},{"sourceId":13816899,"sourceType":"datasetVersion","datasetId":8620533},{"sourceId":272137252,"sourceType":"kernelVersion"},{"sourceId":677607,"sourceType":"modelInstanceVersion","modelInstanceId":513841,"modelId":528480}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"hengck23’s excellent solution (demo submission) - 16.1 baseline:\nhttps://www.kaggle.com/code/hengck23/demo-submission\n\nwasupandceacar’s Net3 pipeline - 17.75 baseline:\nhttps://www.kaggle.com/code/wasupandceacar/physio-v2-3-public\n\nSaner Turhaner ’s Visual QA  - 18.17 baseline:\nhttps://www.kaggle.com/code/sanpier/visual-qa-for-all-stages-of-ecg-digitization","metadata":{}},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"!pip uninstall -y tensorflow\n!uv pip install --no-deps --system --no-index --find-links='/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/setup' connected-components-3d","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2026-01-22T16:11:11.803889Z","iopub.execute_input":"2026-01-22T16:11:11.804406Z","iopub.status.idle":"2026-01-22T16:11:34.155306Z","shell.execute_reply.started":"2026-01-22T16:11:11.804381Z","shell.execute_reply":"2026-01-22T16:11:34.154418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, sys, gc, cv2, numpy as np, pandas as pd, torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.transforms as T\nimport timm\nfrom scipy.signal import savgol_filter\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nsys.path.append('/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet')\nimport stage0_common as s0c\nimport stage1_common as s1c\nimport stage2_common as s2c\nfrom stage0_model import Net as Stage0Net\nfrom stage1_model import Net as Stage1Net\nfrom stage2_model import MyCoordUnetDecoder, encode_with_resnet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:34.157092Z","iopub.execute_input":"2026-01-22T16:11:34.157403Z","iopub.status.idle":"2026-01-22T16:11:46.408440Z","shell.execute_reply.started":"2026-01-22T16:11:34.157382Z","shell.execute_reply":"2026-01-22T16:11:46.407811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Global","metadata":{}},{"cell_type":"code","source":"LEADS_ORDER = [\"I\",\"II\",\"III\",\"aVR\",\"aVL\",\"aVF\",\"V1\",\"V2\",\"V3\",\"V4\",\"V5\",\"V6\"]\nimg_type_list = [\"0001\", \"0003\", \"0004\", \"0005\", \"0006\", \"0009\", \"0010\", \"0011\", \"0012\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.409138Z","iopub.execute_input":"2026-01-22T16:11:46.409576Z","iopub.status.idle":"2026-01-22T16:11:46.413506Z","shell.execute_reply.started":"2026-01-22T16:11:46.409557Z","shell.execute_reply":"2026-01-22T16:11:46.412841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Func","metadata":{}},{"cell_type":"code","source":"def mitigate_shadows(img_bgr: np.ndarray) -> np.ndarray:\n    \"\"\"\n    画像の不均一な影、シミ、シワによる明暗を緩和します。\n    \"\"\"\n    h, w = img_bgr.shape[:2]\n    k_size = min(h, w) // 10\n    if k_size % 2 == 0:\n        k_size += 1\n    \n    if k_size < 3:\n        return img_bgr\n\n    corrected_channels = []\n    \n    for chan in cv2.split(img_bgr):\n        chan_background = cv2.GaussianBlur(chan, (k_size, k_size), 0)\n        chan_bg_safe = cv2.add(chan_background, 1) \n        chan_corrected = cv2.divide(chan, chan_bg_safe, scale=250.0)\n        corrected_channels.append(chan_corrected)\n\n    img_corrected = cv2.merge(corrected_channels)\n    return img_corrected","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.414384Z","iopub.execute_input":"2026-01-22T16:11:46.414724Z","iopub.status.idle":"2026-01-22T16:11:46.427250Z","shell.execute_reply.started":"2026-01-22T16:11:46.414707Z","shell.execute_reply":"2026-01-22T16:11:46.426507Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### preprocess by source","metadata":{}},{"cell_type":"code","source":"def change_color(image_rgb):\n    hsv = cv2.cvtColor(image_rgb, cv2.COLOR_RGB2HSV)\n    h, s, v = cv2.split(hsv)\n    v_denoised = cv2.fastNlMeansDenoising(v, h=5.46)\n    std = np.std(v_denoised)\n    clip_limit = max(1.0, min(3.5, 2.0 + std / 25))\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=(8, 8))\n    v_enhanced = clahe.apply(v_denoised)\n    hsv_enhanced = cv2.merge([h, s, v_enhanced])\n    return cv2.cvtColor(hsv_enhanced, cv2.COLOR_HSV2RGB)\n\ndef change_color_my(image_rgb):\n    hsv = cv2.cvtColor(image_rgb, cv2.COLOR_RGB2HSV)\n    h, s, v = cv2.split(hsv)\n    v_denoised = cv2.fastNlMeansDenoising(v, h=5.5)\n    std = np.std(v_denoised)\n    clip_limit = max(1.0, min(3.5, 2.0 + std / 25))\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=(8, 8))\n    v_enhanced = clahe.apply(v_denoised)\n    hsv_enhanced = cv2.merge([h, s, v_enhanced])\n    return cv2.cvtColor(hsv_enhanced, cv2.COLOR_HSV2RGB)\n\ndef series_dict(series_4row):\n    series_4row = np.asarray(series_4row)\n    if series_4row.ndim == 3: series_4row = series_4row[0]\n    if series_4row.shape[0] != 4 and series_4row.shape[1] == 4: series_4row = series_4row.T\n\n    d = {}\n    names = [\n        ['I','aVR','V1','V4'],\n        ['II_short','aVL','V2','V5'],   \n        ['III','aVF','V3','V6'],\n    ]\n    for r in range(3):\n        for lead, arr in zip(names[r], np.array_split(series_4row[r], 4)):\n            d[lead] = np.asarray(arr, dtype=np.float32)\n\n    d['II'] = np.asarray(series_4row[3], dtype=np.float32)  # full 10s II\n    return d\n\n# ✅ FIX: Einthoven correction on SHORT leads only\ndef dw(d, alpha=0.33):\n    if all(k in d for k in ['I','II_short','III']):\n        L1, L2s, L3 = d['I'], d['II_short'], d['III']\n        e = L2s - (L1 + L3)\n        d['I']        = L1 + alpha*e\n        d['III']      = L3 + alpha*e\n        d['II_short'] = L2s - alpha*e\n    return d\n\ndef clahe_luminance_bgr(img_bgr, clip=2.0, tile=8):\n    lab = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=float(clip), tileGridSize=(int(tile), int(tile)))\n    l2 = clahe.apply(l)\n    return cv2.cvtColor(cv2.merge([l2, a, b]), cv2.COLOR_LAB2BGR)\n\ndef grayworld_white_balance(img_bgr):\n    img = img_bgr.astype(np.float32)\n    b, g, r = cv2.split(img)\n    mb, mg, mr = b.mean(), g.mean(), r.mean()\n    m = (mb + mg + mr) / 3.0\n    b *= (m / (mb + 1e-6)); g *= (m / (mg + 1e-6)); r *= (m / (mr + 1e-6))\n    return np.clip(cv2.merge([b, g, r]), 0, 255).astype(np.uint8)\n\ndef denoise_median(img_bgr, k=3):\n    k = int(k); k = k if k % 2 == 1 else k + 1\n    return cv2.medianBlur(img_bgr, k)\n\ndef denoise_bilateral(img_bgr, d=7, sigmaColor=50, sigmaSpace=50):\n    return cv2.bilateralFilter(img_bgr, d=int(d), sigmaColor=float(sigmaColor), sigmaSpace=float(sigmaSpace))\n\ndef illumination_strength(img_bgr, sigma=35):\n    gray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY).astype(np.float32) / 255.0\n    blur = cv2.GaussianBlur(gray, (0, 0), sigma)\n    return float(np.std(blur))\n\ndef bg_correct_lab_l(img_bgr, k=81):\n    lab = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2LAB)\n    l, a, b = cv2.split(lab)\n    k = int(k); k = k if k % 2 == 1 else k + 1\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (k, k))\n    bg = cv2.morphologyEx(l, cv2.MORPH_OPEN, kernel)\n    l_corr = cv2.subtract(l, bg)\n    l_corr = cv2.normalize(l_corr, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n    return cv2.cvtColor(cv2.merge([l_corr, a, b]), cv2.COLOR_LAB2BGR)\n\ndef preprocess_by_source(img_bgr, source):\n    s = str(source)\n    if s == \"0001\": return img_bgr\n    if s == \"0003\": return clahe_luminance_bgr(grayworld_white_balance(img_bgr), clip=1.2, tile=8)\n    if s == \"0004\": return img_bgr\n    if s == \"0006\":\n        x = denoise_bilateral(img_bgr, d=5, sigmaColor=25, sigmaSpace=25)\n        return clahe_luminance_bgr(x, clip=1.2, tile=8)\n    if s == \"0005\":\n        x = img_bgr\n        if illumination_strength(x, sigma=35) > 0.14: x = bg_correct_lab_l(x, k=81)\n        if cv2.cvtColor(x, cv2.COLOR_BGR2GRAY).std() < 30: x = clahe_luminance_bgr(x, clip=1.1, tile=8)\n        return x\n    if s == \"0009\":\n        x = img_bgr\n        if illumination_strength(x, sigma=35) > 0.14: x = bg_correct_lab_l(x, k=101)\n        return denoise_median(x, k=3)\n    if s == \"0010\":\n        x = img_bgr\n        if illumination_strength(x, sigma=35) > 0.14: x = bg_correct_lab_l(x, k=81)\n        if cv2.cvtColor(x, cv2.COLOR_BGR2GRAY).std() < 30: x = clahe_luminance_bgr(x, clip=1.15, tile=8)\n        return x\n    if s == \"0011\": return clahe_luminance_bgr(grayworld_white_balance(img_bgr), clip=1.2, tile=8)\n    if s == \"0012\": return img_bgr\n    return img_bgr\n\ndef preprocess_by_source_my(img_bgr, source):\n    s = str(source)\n    if s == \"0001\": return img_bgr\n    if s == \"0003\": return clahe_luminance_bgr(grayworld_white_balance(img_bgr), clip=1.2, tile=8)\n    if s == \"0004\": return img_bgr\n    if s == \"0006\":\n        x = img_bgr\n        x = denoise_bilateral(x, d=3, sigmaColor=200, sigmaSpace=25)\n        x = denoise_bilateral(x, d=3, sigmaColor=200, sigmaSpace=25)\n        x = mitigate_shadows(x)\n        return clahe_luminance_bgr(x, clip=1.2, tile=8)        \n    if s == \"0005\":\n        x = img_bgr\n        if illumination_strength(x, sigma=35) > 0.14: x = bg_correct_lab_l(x, k=81)\n        if cv2.cvtColor(x, cv2.COLOR_BGR2GRAY).std() < 30: x = clahe_luminance_bgr(x, clip=1.1, tile=8)\n        return x\n    if s == \"0009\":\n        x = img_bgr\n        if illumination_strength(x, sigma=35) > 0.14: x = bg_correct_lab_l(x, k=81)\n        if cv2.cvtColor(x, cv2.COLOR_BGR2GRAY).std() < 30: x = clahe_luminance_bgr(x, clip=1.15, tile=8)\n        return x\n    if s == \"0010\":\n        x = img_bgr\n        if illumination_strength(x, sigma=35) > 0.14: x = bg_correct_lab_l(x, k=81)\n        if cv2.cvtColor(x, cv2.COLOR_BGR2GRAY).std() < 30: x = clahe_luminance_bgr(x, clip=1.15, tile=8)\n        return x\n    if s == \"0011\": return clahe_luminance_bgr(grayworld_white_balance(img_bgr), clip=1.2, tile=8)\n    if s == \"0012\": return img_bgr\n    return img_bgr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.428966Z","iopub.execute_input":"2026-01-22T16:11:46.429246Z","iopub.status.idle":"2026-01-22T16:11:46.452279Z","shell.execute_reply.started":"2026-01-22T16:11:46.429227Z","shell.execute_reply":"2026-01-22T16:11:46.451537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Metric\nfrom typing import Tuple\n\nimport numpy as np\nimport pandas as pd\n\nimport scipy.optimize\nimport scipy.signal\n\n\nLEADS = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\nMAX_TIME_SHIFT = 0.2\nPERFECT_SCORE = 384\n\n\nclass ParticipantVisibleError(Exception):\n    pass\n\n\ndef compute_power(label: np.ndarray, prediction: np.ndarray) -> Tuple[float, float]:\n    if label.ndim != 1 or prediction.ndim != 1:\n        raise ParticipantVisibleError('Inputs must be 1-dimensional arrays.')\n    finite_mask = np.isfinite(prediction)\n    if not np.any(finite_mask):\n        raise ParticipantVisibleError(\"The 'prediction' array contains no finite values (all NaN or inf).\")\n\n    prediction[~np.isfinite(prediction)] = 0\n    noise = label - prediction\n    p_signal = np.sum(label**2)\n    p_noise = np.sum(noise**2)\n    return p_signal, p_noise\n\n\ndef compute_snr(signal: float, noise: float) -> float:\n    if noise == 0:\n        # Perfect reconstruction\n        snr = PERFECT_SCORE\n    elif signal == 0:\n        snr = 0\n    else:\n        snr = min((signal / noise), PERFECT_SCORE)\n    return snr\n\n\ndef align_signals(label: np.ndarray, pred: np.ndarray, max_shift: float = float('inf')) -> np.ndarray:\n    if np.any(~np.isfinite(label)):\n        raise ParticipantVisibleError('values in label should all be finite')\n    if np.sum(np.isfinite(pred)) == 0:\n        raise ParticipantVisibleError('prediction can not all be infinite')\n\n    # Initialize the reference and digitized signals.\n    label_arr = np.asarray(label, dtype=np.float64)\n    pred_arr = np.asarray(pred, dtype=np.float64)\n\n    label_mean = np.mean(label_arr)\n    pred_mean = np.mean(pred_arr)\n\n    label_arr_centered = label_arr - label_mean\n    pred_arr_centered = pred_arr - pred_mean\n\n    # Compute the correlation between the reference and digitized signals and locate the maximum correlation.\n    correlation = scipy.signal.correlate(label_arr_centered, pred_arr_centered, mode='full')\n\n    n_label = np.size(label_arr)\n    n_pred = np.size(pred_arr)\n\n    lags = scipy.signal.correlation_lags(n_label, n_pred, mode='full')\n    valid_lags_mask = (lags >= -max_shift) & (lags <= max_shift)\n\n    max_correlation = np.nanmax(correlation[valid_lags_mask])\n    all_max_indices = np.flatnonzero(correlation == max_correlation)\n    best_idx = min(all_max_indices, key=lambda i: abs(lags[i]))\n    time_shift = lags[best_idx]\n    start_padding_len = max(time_shift, 0)\n    pred_slice_start = max(-time_shift, 0)\n    pred_slice_end = min(n_label - time_shift, n_pred)\n    end_padding_len = max(n_label - n_pred - time_shift, 0)\n    aligned_pred = np.concatenate((np.full(start_padding_len, np.nan), pred_arr[pred_slice_start:pred_slice_end], np.full(end_padding_len, np.nan)))\n\n    def objective_func(v_shift):\n        return np.nansum((label_arr - (aligned_pred - v_shift)) ** 2)\n\n    if np.any(np.isfinite(label_arr) & np.isfinite(aligned_pred)):\n        results = scipy.optimize.minimize_scalar(objective_func, method='Brent')\n        vertical_shift = results.x\n        aligned_pred -= vertical_shift\n    return aligned_pred\n\n\ndef _calculate_image_score(group: pd.DataFrame) -> float:\n    \"\"\"Helper function to calculate the total SNR score for a single image group.\"\"\"\n\n    unique_fs_values = group['fs'].unique()\n    if len(unique_fs_values) != 1:\n        raise ParticipantVisibleError('Sampling frequency should be consistent across each ecg')\n    sampling_frequency = unique_fs_values[0]\n    if sampling_frequency != int(len(group[group['lead'] == 'II']) / 10):\n        raise ParticipantVisibleError('The sequence_length should be sampling frequency * 10s')\n    sum_signal = 0\n    sum_noise = 0\n    for lead in LEADS:\n        sub = group[group['lead'] == lead]\n        label = sub['value_true'].values\n        pred = sub['value_pred'].values\n\n        aligned_pred = align_signals(label, pred, int(sampling_frequency * MAX_TIME_SHIFT))\n        p_signal, p_noise = compute_power(label, aligned_pred)\n        sum_signal += p_signal\n        sum_noise += p_noise\n    return compute_snr(sum_signal, sum_noise)\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    \"\"\"\n    Compute the mean Signal-to-Noise Ratio (SNR) across multiple ECG leads and images for the PhysioNet 2025 competition.\n    The final score is the average of the sum of SNRs over different lines, averaged over all unique images.\n    Args:\n        solution: DataFrame with ground truth values. Expected columns: 'id' and one for each lead.\n        submission: DataFrame with predicted values. Expected columns: 'id' and one for each lead.\n        row_id_column_name: The name of the unique identifier column, typically 'id'.\n    Returns:\n        The final competition score.\n\n    Examples\n    --------\n    >>> import pandas as pd\n    >>> import numpy as np\n    >>> row_id_column_name = \"id\"\n    >>> solution = pd.DataFrame({'id': ['343_0_I', '343_1_I', '343_2_I', '343_0_III', '343_1_III','343_2_III','343_0_aVR', '343_1_aVR','343_2_aVR',\\\n    '343_0_aVL', '343_1_aVL', '343_2_aVL', '343_0_aVF', '343_1_aVF','343_2_aVF','343_0_V1', '343_1_V1', '343_2_V1','343_0_V2', '343_1_V2','343_2_V2',\\\n    '343_0_V3', '343_1_V3', '343_2_V3','343_0_V4', '343_1_V4', '343_2_V4', '343_0_V5', '343_1_V5','343_2_V5','343_0_V6', '343_1_V6','343_2_V6',\\\n    '343_0_II', '343_1_II','343_2_II', '343_3_II', '343_4_II', '343_5_II','343_6_II', '343_7_II','343_8_II','343_9_II','343_10_II','343_11_II'],\\\n    'fs': [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1],\\\n    'value':[0.1,0.3,0.4,0.6,0.6,0.4,0.2,0.3,0.4,0.5,0.2,0.7,0.2,0.3,0.4,0.8,0.6,0.7, 0.2,0.3,-0.1,0.5,0.6,0.7,0.2,0.9,0.4,0.5,0.6,0.7,0.1,0.3,0.4,\\\n    0.6,0.6,0.4,0.2,0.3,0.4,0.5,0.2,0.7,0.2,0.3,0.4]})\n    >>> submission = solution.copy()\n    >>> round(score(solution, submission, row_id_column_name), 4)\n    25.8433\n    >>> submission.loc[0, 'value'] = 0.9 # Introduce some noise\n    >>> round(score(solution, submission, row_id_column_name), 4)\n    13.6291\n    >>> submission.loc[4, 'value'] = 0.3 # Introduce some noise\n    >>> round(score(solution, submission, row_id_column_name), 4)\n    13.0576\n\n    >>> solution = pd.DataFrame({'id': ['343_0_I', '343_1_I', '343_2_I', '343_0_III', '343_1_III','343_2_III','343_0_aVR', '343_1_aVR','343_2_aVR',\\\n    '343_0_aVL', '343_1_aVL', '343_2_aVL', '343_0_aVF', '343_1_aVF','343_2_aVF','343_0_V1', '343_1_V1', '343_2_V1','343_0_V2', '343_1_V2','343_2_V2',\\\n    '343_0_V3', '343_1_V3', '343_2_V3','343_0_V4', '343_1_V4', '343_2_V4', '343_0_V5', '343_1_V5','343_2_V5','343_0_V6', '343_1_V6','343_2_V6',\\\n    '343_0_II', '343_1_II','343_2_II', '343_3_II', '343_4_II', '343_5_II','343_6_II', '343_7_II','343_8_II','343_9_II','343_10_II','343_11_II'],\\\n    'fs': [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1],\\\n    'value':[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0]})\n    >>> round(score(solution, submission, row_id_column_name), 4)\n    -384\n    >>> submission = solution.copy()\n    >>> round(score(solution, submission, row_id_column_name), 4)\n    25.8433\n\n    >>> # test alignment\n    >>> label = np.array([0, 1, 2, 1, 0])\n    >>> pred = np.array([0, 1, 2, 1, 0])\n    >>> aligned = align_signals(label, pred)\n    >>> expected_array = np.array([0, 1, 2, 1, 0])\n    >>> np.allclose(aligned, expected_array, equal_nan=True)\n    True\n\n    >>> # Test 2: Vertical shift (DC offset) should be removed\n    >>> label = np.array([0, 1, 2, 1, 0])\n    >>> pred = np.array([10, 11, 12, 11, 10])\n    >>> aligned = align_signals(label, pred)\n    >>> expected_array = np.array([0, 1, 2, 1, 0])\n    >>> np.allclose(aligned, expected_array, equal_nan=True)\n    True\n\n    >>> # Test 3: Time shift should be corrected\n    >>> label = np.array([0, 0, 1, 2, 1, 0., 0.])\n    >>> pred = np.array([1, 2, 1, 0, 0, 0, 0])\n    >>> aligned = align_signals(label, pred)\n    >>> expected_array = np.array([np.nan, np.nan, 1, 2, 1, 0, 0])\n    >>> np.allclose(aligned, expected_array, equal_nan=True)\n    True\n    \n    >>> # Test 4: max_shift constraint prevents optimal alignment\n    >>> label = np.array([0, 0, 0, 0, 1, 2, 1]) # Peak is far\n    >>> pred = np.array([1, 2, 1, 0, 0, 0, 0])\n    >>> aligned = align_signals(label, pred, max_shift=10)\n    >>> expected_array = np.array([ np.nan, np.nan, np.nan, np.nan, 1, 2, 1])\n    >>> np.allclose(aligned, expected_array, equal_nan=True)\n    True\n\n    \"\"\"\n    for df in [solution, submission]:\n        if row_id_column_name not in df.columns:\n            raise ParticipantVisibleError(f\"'{row_id_column_name}' column not found in DataFrame.\")\n        if df['value'].isna().any():\n            raise ParticipantVisibleError('NaN exists in solution/submission')\n        if not np.isfinite(df['value']).all():\n            raise ParticipantVisibleError('Infinity exists in solution/submission')\n\n    submission = submission[['id', 'value']]\n    merged_df = pd.merge(solution, submission, on=row_id_column_name, suffixes=('_true', '_pred'))\n    merged_df['image_id'] = merged_df[row_id_column_name].str.split('_').str[0]\n    merged_df['row_id'] = merged_df[row_id_column_name].str.split('_').str[1].astype('int64')\n    merged_df['lead'] = merged_df[row_id_column_name].str.split('_').str[2]\n    merged_df.sort_values(by=['image_id', 'row_id', 'lead'], inplace=True)\n    image_scores = merged_df.groupby('image_id').apply(_calculate_image_score, include_groups=False)\n    return max(float(10 * np.log10(image_scores.mean())), -PERFECT_SCORE)","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2026-01-22T16:11:46.452921Z","iopub.execute_input":"2026-01-22T16:11:46.453123Z","iopub.status.idle":"2026-01-22T16:11:46.470733Z","shell.execute_reply.started":"2026-01-22T16:11:46.453082Z","shell.execute_reply":"2026-01-22T16:11:46.469972Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluate Stage 1","metadata":{}},{"cell_type":"code","source":"def stage1_quality(s1_rgb):\n    g = cv2.cvtColor(s1_rgb.astype(np.uint8), cv2.COLOR_RGB2GRAY)\n    e = cv2.Canny(g, 50, 150)\n    density = e.mean() / 255.0\n    gx = cv2.Sobel(g, cv2.CV_32F, 1, 0, ksize=3)\n    gy = cv2.Sobel(g, cv2.CV_32F, 0, 1, ksize=3)\n    ax = float(np.mean(np.abs(gx))); ay = float(np.mean(np.abs(gy)))\n    anis = max(ax, ay) / (min(ax, ay) + 1e-6)\n    return float(density * 0.7 + np.tanh(anis - 1.0) * 0.3)","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2026-01-22T16:11:46.471508Z","iopub.execute_input":"2026-01-22T16:11:46.471728Z","iopub.status.idle":"2026-01-22T16:11:46.484603Z","shell.execute_reply.started":"2026-01-22T16:11:46.471712Z","shell.execute_reply":"2026-01-22T16:11:46.483911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Pipeline","metadata":{}},{"cell_type":"code","source":"class Net3(nn.Module):\n    def __init__(self, pretrained=True):\n        super().__init__()\n        encoder_dim = [64, 128, 256, 512]\n        decoder_dim = [128, 64, 32, 16]\n        self.encoder = timm.create_model(\n            model_name='resnet34.a3_in1k',\n            pretrained=pretrained,\n            in_chans=3,\n            num_classes=0,\n            global_pool=''\n        )\n        self.decoder = MyCoordUnetDecoder(\n            in_channel=encoder_dim[-1],\n            skip_channel=encoder_dim[:-1][::-1] + [0],\n            out_channel=decoder_dim,\n            scale=[2, 2, 2, 2]\n        )\n        self.pixel = nn.Conv2d(decoder_dim[-1], 4, 1)\n\n    def forward(self, image):\n        encode = encode_with_resnet(self.encoder, image)\n        last, _ = self.decoder(feature=encode[-1], skip=encode[:-1][::-1] + [None])\n        return self.pixel(last)","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2026-01-22T16:11:46.485243Z","iopub.execute_input":"2026-01-22T16:11:46.485416Z","iopub.status.idle":"2026-01-22T16:11:46.492467Z","shell.execute_reply.started":"2026-01-22T16:11:46.485401Z","shell.execute_reply":"2026-01-22T16:11:46.491790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PhysioPipeline:\n    def __init__(self, device=\"cuda:0\"):\n        self.device = device\n        self.stage0_net = self.stage1_net = self.stage2_net = None\n        self.x0, self.x1 = 0, 2176\n        self.y0, self.y1 = 0, 1696\n        self.zero_mv = [703.5, 987.5, 1271.5, 1531.5]\n        self.mv_to_pixel = 78.8\n        self.t0, self.t1 = 235, 4161\n        self.resize = T.Resize((1696, 4352), interpolation=T.InterpolationMode.BILINEAR)\n\n    def load_models(self, stage0_w, stage1_w, stage2_w):\n        self.stage0_net = s0c.load_net(Stage0Net(pretrained=False), stage0_w).to(self.device).eval()\n        self.stage1_net = s1c.load_net(Stage1Net(pretrained=False), stage1_w).to(self.device).eval()\n        self.stage2_net = Net3(pretrained=False).to(self.device).eval()\n        st = torch.load(stage2_w, map_location=\"cpu\")\n        if isinstance(st, dict) and \"state_dict\" in st: st = st[\"state_dict\"]\n        self.stage2_net.load_state_dict(st, strict=True)\n\n    def run_stage0(self, img_bgr):\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        img_for_model = change_color(img_rgb)\n        batch = s0c.image_to_batch(img_for_model)\n        with torch.no_grad(), torch.amp.autocast(self.device.split(\":\")[0], dtype=torch.float32):\n            output = self.stage0_net(batch)\n        rotated, keypoint = s0c.output_to_predict(img_rgb, batch, output)\n        normalised, _, _ = s0c.normalise_by_homography(rotated, keypoint)\n\n        return normalised\n\n    def run_stage0_my(self, img_bgr):\n        img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n        img_for_model = change_color_my(img_rgb)\n        batch = s0c.image_to_batch(img_for_model)\n        with torch.no_grad(), torch.amp.autocast(self.device.split(\":\")[0], dtype=torch.float32):\n            output = self.stage0_net(batch)\n        rotated, keypoint = s0c.output_to_predict(img_rgb, batch, output)\n        normalised, _, _ = s0c.normalise_by_homography(rotated, keypoint)\n\n        return normalised\n\n    def run_stage1(self, stage0_img_rgb):\n        image = stage0_img_rgb\n        batch = {'image': torch.from_numpy(np.ascontiguousarray(image.transpose(2, 0, 1))).unsqueeze(0)}\n        with torch.no_grad(), torch.amp.autocast(self.device.split(\":\")[0], dtype=torch.float32):\n            output = self.stage1_net(batch)\n        gridpoint_xy, _ = s1c.output_to_predict(image, batch, output)\n        rectified = s1c.rectify_image(image, gridpoint_xy)\n\n        return rectified\n\n    def run_stage2(self, stage1_img_rgb, length):\n        img = stage1_img_rgb[self.y0:self.y1, self.x0:self.x1] / 255.0\n        batch = self.resize(torch.from_numpy(np.ascontiguousarray(img.transpose(2, 0, 1))).unsqueeze(0)).float().to(self.device)\n        with torch.no_grad(), torch.amp.autocast(self.device.split(\":\")[0], dtype=torch.float32):\n            output = self.stage2_net(batch)\n        pixel = torch.sigmoid(output).float().cpu().numpy()[0]\n        series_in_pixel = s2c.pixel_to_series(pixel[..., self.t0:self.t1], self.zero_mv, length)\n        series = (np.array(self.zero_mv).reshape(4, 1) - series_in_pixel) / self.mv_to_pixel\n        for i in range(4):\n            series[i] = savgol_filter(series[i], window_length=7, polyorder=2)\n            \n        return series","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.493183Z","iopub.execute_input":"2026-01-22T16:11:46.493421Z","iopub.status.idle":"2026-01-22T16:11:46.516250Z","shell.execute_reply.started":"2026-01-22T16:11:46.493406Z","shell.execute_reply":"2026-01-22T16:11:46.515740Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CLS_MODEL_NAME=\"efficientnet_b2\"\nCLS_NUM_CLASSES=12\nCLS_RESOLUTION=256\nCLS_CKPT_PATH=\"/kaggle/input/physionet-image-multi-class-train/efficientnet_b2_full_train.pth\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.517007Z","iopub.execute_input":"2026-01-22T16:11:46.517249Z","iopub.status.idle":"2026-01-22T16:11:46.530423Z","shell.execute_reply.started":"2026-01-22T16:11:46.517234Z","shell.execute_reply":"2026-01-22T16:11:46.529696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_classifier(device=\"cuda\"):\n    m = timm.create_model(CLS_MODEL_NAME, pretrained=False, num_classes=CLS_NUM_CLASSES)\n    st = torch.load(CLS_CKPT_PATH, map_location=\"cpu\")\n    if isinstance(st, dict) and \"state_dict\" in st: st = st[\"state_dict\"]\n    st2 = {k.replace(\"module.\",\"\"): v for k, v in st.items()}\n    m.load_state_dict(st2, strict=False)\n    return m.to(device).eval()\n\ndef cls_preprocess_bgr(img_bgr):\n    img = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (CLS_RESOLUTION, CLS_RESOLUTION), interpolation=cv2.INTER_AREA).astype(np.float32)/255.0\n    mean = np.array([0.485,0.456,0.406], np.float32); std = np.array([0.229,0.224,0.225], np.float32)\n    img = (img - mean) / std\n    return torch.from_numpy(img).permute(2,0,1).unsqueeze(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.531414Z","iopub.execute_input":"2026-01-22T16:11:46.531922Z","iopub.status.idle":"2026-01-22T16:11:46.543207Z","shell.execute_reply.started":"2026-01-22T16:11:46.531898Z","shell.execute_reply":"2026-01-22T16:11:46.542493Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef predict_source_suffix(model, img_bgr, device=\"cuda\"):\n    x = cls_preprocess_bgr(img_bgr).to(device)\n    p = F.softmax(model(x), dim=1)[0]\n    cls = int(torch.argmax(p).item())\n    return f\"{cls+1:04d}\"\n\ndef pseudo_colorize_grid(\n    img_bgr: np.ndarray, \n    kernel_size: int = 5, \n    intensity_threshold: int = 80\n) -> np.ndarray:\n    \"\"\"\n    太さ（モルフォロジー）と濃さ（輝度閾値）の両方を用いて信号線を特定し、\n    それ以外のグリッド成分を赤色化する。\n    \n    Args:\n        img_bgr: 入力画像（BGR 3チャンネル）\n        kernel_size: 太い線を特定するためのカーネルサイズ\n        intensity_threshold: 信号線とみなす暗さの閾値\n    \"\"\"\n    # 1. グレースケール変換\n    gray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\n    \n    # --- 2. 信号線マスクの作成 (太さ + 濃さ) ---\n    # (A) 太さによる抽出: モルフォロジー演算 (オープニング)\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (kernel_size, kernel_size))\n    inv = cv2.bitwise_not(gray)\n    signal_thick = cv2.morphologyEx(inv, cv2.MORPH_OPEN, kernel)\n    \n    # (B) 濃さによる抽出: 輝度値が intensity_threshold より低い（黒い）画素を抽出\n    _, signal_dark = cv2.threshold(inv, 255 - intensity_threshold, 255, cv2.THRESH_BINARY)\n    \n    # (A)と(B)の論理和(OR)をとり、太い線または濃い線のいずれかを信号線とする\n    signal_mask = cv2.bitwise_or(signal_thick, signal_dark)\n    \n    # --- 3. 背景（グリッド）の赤色化処理 ---\n    res_r = np.full_like(gray, 255)\n    res_g = gray.copy()\n    res_b = gray.copy()\n    img_red_bg_rgb = cv2.merge([res_r, res_g, res_b])\n\n    # --- 4. 合成 ---\n    img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n    mask_float = signal_mask.astype(np.float32) / 255.0\n    mask_3ch = cv2.merge([mask_float, mask_float, mask_float])\n    \n    result = (img_rgb.astype(np.float32) * mask_3ch + \n              img_red_bg_rgb.astype(np.float32) * (1.0 - mask_3ch))\n    \n    return result.astype(np.uint8)\n    \ndef select_stage1_with_source(pipeline, img_raw_bgr, pred_source_suffix, selector_margin=1.02):\n    img_pp = preprocess_by_source_my(img_raw_bgr.copy(), pred_source_suffix)\n    s1_raw = pipeline.run_stage1(pipeline.run_stage0(img_raw_bgr))\n    if pred_source_suffix!=\"0004\" and pred_source_suffix!=\"0012\":\n        s1_pp = pipeline.run_stage1(pipeline.run_stage0(img_pp))\n\n    \n    if pred_source_suffix==\"0004\" or pred_source_suffix==\"0012\":\n        s1_raw = pseudo_colorize_grid(s1_raw)\n        s1_pp = s1_raw.copy()\n\n    \n    q_raw  = stage1_quality(s1_raw)\n    q_pp   = stage1_quality(s1_pp)\n\n    if (q_pp > q_raw * selector_margin) or (pred_source_suffix==\"0006\"):\n        return s1_pp\n    else:\n        return s1_raw\n\ndef make_submission_from_pred(base_id: str, fs: int, sig_len: int, d_series: dict):\n    base_id = str(base_id); fs = int(fs); sig_len = int(sig_len)\n    n_short = int(np.floor(fs * 2.5))\n\n    def take_segment(y, n):\n        y = np.asarray(y, dtype=np.float64)\n        if len(y) >= n: return y[:n]\n        if len(y) == 0: return np.zeros(n, np.float64)\n        return np.concatenate([y, np.full(n - len(y), y[-1], np.float64)])\n\n    rows = []\n    for lead in LEADS_ORDER:\n        y = np.asarray(d_series[lead], dtype=np.float64)\n        seg = take_segment(y, sig_len if lead==\"II\" else n_short)\n        rows.append(pd.DataFrame({\"id\":[f\"{base_id}_{i}_{lead}\" for i in range(len(seg))],\n                                  \"value\":seg.astype(np.float32)}))\n    return pd.concat(rows, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.544117Z","iopub.execute_input":"2026-01-22T16:11:46.544314Z","iopub.status.idle":"2026-01-22T16:11:46.555698Z","shell.execute_reply.started":"2026-01-22T16:11:46.544299Z","shell.execute_reply":"2026-01-22T16:11:46.554899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Runnning One","metadata":{}},{"cell_type":"code","source":"def show_image(image):\n    plt.figure(figsize=(10, 10))\n    plt.imshow(image)\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:11:46.556617Z","iopub.execute_input":"2026-01-22T16:11:46.556913Z","iopub.status.idle":"2026-01-22T16:11:46.567311Z","shell.execute_reply.started":"2026-01-22T16:11:46.556889Z","shell.execute_reply":"2026-01-22T16:11:46.566744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_id = \"1006427285\"\nimg_type = \"0006\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:12:44.912632Z","iopub.execute_input":"2026-01-22T16:12:44.912913Z","iopub.status.idle":"2026-01-22T16:12:44.916631Z","shell.execute_reply.started":"2026-01-22T16:12:44.912891Z","shell.execute_reply":"2026-01-22T16:12:44.915837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"WORK_DIR=\"/kaggle/input/physionet-ecg-image-digitization\"\ndf_train = pd.read_csv(f\"{WORK_DIR}/train.csv\")\ndf_train[\"id\"] = df_train[\"id\"].astype(str)\nsample_submission = pd.read_parquet(f\"{WORK_DIR}/sample_submission.parquet\")[[\"id\"]]\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\npipeline = PhysioPipeline(device=\"cuda:0\" if device==\"cuda\" else \"cpu\")\npipeline.load_models(\n    stage0_w=\"/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth\",\n    stage1_w=\"/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth\",\n    stage2_w=\"/kaggle/input/physio-seg-public/pytorch/net3_009_4200/1/iter_0004200.pt\",\n)\ncls_model = build_classifier(device=device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:12:47.217306Z","iopub.execute_input":"2026-01-22T16:12:47.218152Z","iopub.status.idle":"2026-01-22T16:12:48.700000Z","shell.execute_reply.started":"2026-01-22T16:12:47.218114Z","shell.execute_reply":"2026-01-22T16:12:48.699380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res = []\nimg_path = f\"{WORK_DIR}/train/{img_id}/{img_id}-{img_type}.png\"\nimg_raw = cv2.imread(img_path, cv2.IMREAD_COLOR)\nif img_raw is None:\n    raise FileNotFoundError(img_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:12:48.701625Z","iopub.execute_input":"2026-01-22T16:12:48.701938Z","iopub.status.idle":"2026-01-22T16:12:49.605555Z","shell.execute_reply.started":"2026-01-22T16:12:48.701921Z","shell.execute_reply":"2026-01-22T16:12:49.604675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fs = df_train[(df_train['id']==img_id)].iloc[0].fs\n# sig_len = df_train[(df_train['id']==img_id)].iloc[0].sig_len\n\n# pred_src = predict_source_suffix(cls_model, img_raw, device=device)\n# s1 = select_stage1_with_source(pipeline, img_raw, pred_src, selector_margin=1.02)\n# show_image(s1)\n\n# # series_4row = pipeline.run_stage2(s1, length=sig_len)\n\n# # d = dw(series_dict(series_4row))         # ✅ now safe (uses II_short)\n# # res.append(make_submission_from_pred(img_id, fs, sig_len, d))\n# # gc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T16:12:49.606406Z","iopub.execute_input":"2026-01-22T16:12:49.606675Z","iopub.status.idle":"2026-01-22T16:13:14.833003Z","shell.execute_reply.started":"2026-01-22T16:12:49.606649Z","shell.execute_reply":"2026-01-22T16:13:14.831999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# s1_raw = pipeline.run_stage1(pipeline.run_stage0(img_raw)) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T01:21:39.884150Z","iopub.execute_input":"2026-01-22T01:21:39.884518Z","iopub.status.idle":"2026-01-22T01:21:49.889391Z","shell.execute_reply.started":"2026-01-22T01:21:39.884494Z","shell.execute_reply":"2026-01-22T01:21:49.888722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# img_pp = preprocess_by_source_my(img_raw.copy(), \"0006\")\n# s1_pp  = pipeline.run_stage1(pipeline.run_stage0(img_pp))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T01:25:30.746461Z","iopub.execute_input":"2026-01-22T01:25:30.746939Z","iopub.status.idle":"2026-01-22T01:25:44.578566Z","shell.execute_reply.started":"2026-01-22T01:25:30.746917Z","shell.execute_reply":"2026-01-22T01:25:44.577708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show_image(s1_pp)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T01:25:44.579795Z","iopub.execute_input":"2026-01-22T01:25:44.580087Z","iopub.status.idle":"2026-01-22T01:25:45.231028Z","shell.execute_reply.started":"2026-01-22T01:25:44.580063Z","shell.execute_reply":"2026-01-22T01:25:45.230289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# q_raw  = stage1_quality(s1_raw)\n# q_pp   = stage1_quality(s1_pp)\n# print(f\"q_raw: {q_raw}, q_pp: {q_pp}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T01:24:30.200975Z","iopub.execute_input":"2026-01-22T01:24:30.201497Z","iopub.status.idle":"2026-01-22T01:24:30.322391Z","shell.execute_reply.started":"2026-01-22T01:24:30.201475Z","shell.execute_reply":"2026-01-22T01:24:30.321667Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Running Test","metadata":{}},{"cell_type":"code","source":"WORK_DIR=\"/kaggle/input/physionet-ecg-image-digitization\"\ndf_test = pd.read_csv(f\"{WORK_DIR}/test.csv\")\ndf_test[\"id\"] = df_test[\"id\"].astype(str)\nsample_submission = pd.read_parquet(f\"{WORK_DIR}/sample_submission.parquet\")[[\"id\"]]\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\npipeline = PhysioPipeline(device=\"cuda:0\" if device==\"cuda\" else \"cpu\")\npipeline.load_models(\n    stage0_w=\"/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth\",\n    stage1_w=\"/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth\",\n    stage2_w=\"/kaggle/input/physio-seg-public/pytorch/net3_009_4200/1/iter_0004200.pt\",\n)\ncls_model = build_classifier(device=device)\n\nres = []\nfor sample_id, g in df_test.groupby(\"id\", sort=True):\n    img_path = f\"{WORK_DIR}/test/{sample_id}.png\"\n    img_raw = cv2.imread(img_path, cv2.IMREAD_COLOR)\n    if img_raw is None:\n        raise FileNotFoundError(img_path)\n\n    fs = int(g.fs.iloc[0])\n    sig_len = int(g.loc[g.lead==\"II\",\"number_of_rows\"].iloc[0])\n\n    pred_src = predict_source_suffix(cls_model, img_raw, device=device)\n    s1 = select_stage1_with_source(pipeline, img_raw, pred_src, selector_margin=1.02)\n    series_4row = pipeline.run_stage2(s1, length=sig_len)\n\n    d = dw(series_dict(series_4row))         # ✅ now safe (uses II_short)\n    res.append(make_submission_from_pred(sample_id, fs, sig_len, d))\n    gc.collect()\n\ndf_submission = pd.concat(res, ignore_index=True)\ndf_submission = df_submission.set_index(\"id\").reindex(sample_submission[\"id\"]).reset_index()\n\nassert df_submission[\"value\"].notna().all()\nassert (df_submission[\"id\"].values == sample_submission[\"id\"].values).all()\n\ndf_submission.to_csv(\"submission.csv\", index=False)\nprint(\"OK  submission.csv\", df_submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-18T09:49:52.205849Z","iopub.execute_input":"2026-01-18T09:49:52.206061Z","iopub.status.idle":"2026-01-18T09:50:22.923114Z","shell.execute_reply.started":"2026-01-18T09:49:52.206046Z","shell.execute_reply":"2026-01-18T09:50:22.922395Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Running Train","metadata":{}},{"cell_type":"code","source":"img_type_list = [\"0006\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:43:18.653374Z","iopub.execute_input":"2026-01-22T12:43:18.653634Z","iopub.status.idle":"2026-01-22T12:43:18.657209Z","shell.execute_reply.started":"2026-01-22T12:43:18.653615Z","shell.execute_reply":"2026-01-22T12:43:18.656445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_train = 10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:43:18.658347Z","iopub.execute_input":"2026-01-22T12:43:18.658526Z","iopub.status.idle":"2026-01-22T12:43:18.671398Z","shell.execute_reply.started":"2026-01-22T12:43:18.658513Z","shell.execute_reply":"2026-01-22T12:43:18.670751Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"WORK_DIR=\"/kaggle/input/physionet-ecg-image-digitization\"\ndf_train = pd.read_csv(f\"{WORK_DIR}/train.csv\")\ndf_train[\"id\"] = df_train[\"id\"].astype(str)\nsample_submission = pd.read_parquet(f\"{WORK_DIR}/sample_submission.parquet\")[[\"id\"]]\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\npipeline = PhysioPipeline(device=\"cuda:0\" if device==\"cuda\" else \"cpu\")\npipeline.load_models(\n    stage0_w=\"/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage0-last.checkpoint.pth\",\n    stage1_w=\"/kaggle/input/hengck23-submit-physionet/hengck23-submit-physionet/weight/stage1-last.checkpoint.pth\",\n    stage2_w=\"/kaggle/input/physio-seg-public/pytorch/net3_009_4200/1/iter_0004200.pt\",\n)\ncls_model = build_classifier(device=device)\n\nfor img_type in img_type_list:\n    res = []\n    for i, (sample_id, g) in enumerate(df_train.groupby(\"id\", sort=False)):\n        if i >= num_train:\n            break\n    \n        img_path = f\"{WORK_DIR}/train/{sample_id}/{sample_id}-{img_type}.png\"\n        img_raw = cv2.imread(img_path, cv2.IMREAD_COLOR)\n        if img_raw is None:\n            raise FileNotFoundError(img_path)\n    \n        fs = int(g.fs.iloc[0])\n        sig_len = g.iloc[0].sig_len\n    \n        s1 = select_stage1_with_source(pipeline, img_raw, img_type, selector_margin=1.02)\n        series_4row = pipeline.run_stage2(s1, length=sig_len)\n    \n        d = dw(series_dict(series_4row))         # ✅ now safe (uses II_short)\n        res.append(make_submission_from_pred(sample_id, fs, sig_len, d))\n        gc.collect()\n\n    submission_df_train = pd.concat(res, ignore_index=True)\n    submission_df_train.to_csv(f'train_submission_{img_type}.csv', index=False)\n    print(f\"{img_type} done!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:43:18.672336Z","iopub.execute_input":"2026-01-22T12:43:18.672586Z","iopub.status.idle":"2026-01-22T12:47:57.394978Z","shell.execute_reply.started":"2026-01-22T12:43:18.672571Z","shell.execute_reply":"2026-01-22T12:47:57.394158Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:47:57.396603Z","iopub.execute_input":"2026-01-22T12:47:57.397226Z","iopub.status.idle":"2026-01-22T12:47:57.416091Z","shell.execute_reply.started":"2026-01-22T12:47:57.397203Z","shell.execute_reply":"2026-01-22T12:47:57.415491Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_solution_df(train, train_dir, target_id) -> list:\n    csv_path = os.path.join(train_dir, str(target_id), f\"{target_id}.csv\")\n    df = pd.read_csv(csv_path)\n    \n    # ---- 誘導リスト（CSVに含まれている列を使用） ----\n    leads = [col for col in df.columns if col in LEADS]\n    solution_data = []\n    \n    # ---- 各リードの信号を展開 ----\n    for lead in leads:\n        # 信号読み込み（NaNを含む）\n        signal_raw = df[lead].values.astype(np.float32)\n        signal_true = signal_raw[~np.isnan(signal_raw)]\n    \n        # ---- 長さを指定してリサンプリング ----\n        fs = train.loc[train[\"id\"] == target_id, \"fs\"].values[0]\n        target_len = int(fs*10) if lead == 'II' else int(fs*2.5)\n        signal_resampled = resample(signal_true, target_len)\n    \n        # 各サンプルごとに1行ずつ記録\n        for j, v_true in enumerate(signal_resampled):\n            row_id = f\"{target_id}_{j}_{lead}\"\n            solution_data.append({'id': row_id, 'value': float(v_true), 'fs': fs})\n\n    return solution_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:47:57.416742Z","iopub.execute_input":"2026-01-22T12:47:57.416970Z","iopub.status.idle":"2026-01-22T12:47:57.422985Z","shell.execute_reply.started":"2026-01-22T12:47:57.416953Z","shell.execute_reply":"2026-01-22T12:47:57.422189Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nfrom scipy.signal import resample\ntrain_dir = Path('/kaggle/input/physionet-ecg-image-digitization/train/')\n\nall_solution_data = []\nfor i, row in df_train.iterrows():\n    if i >= num_train:\n        break\n    target_id = row[\"id\"]\n    all_solution_data.extend(make_solution_df(df_train, train_dir, target_id))\n\nsolution_df = pd.DataFrame(all_solution_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:47:57.423771Z","iopub.execute_input":"2026-01-22T12:47:57.424615Z","iopub.status.idle":"2026-01-22T12:47:58.211941Z","shell.execute_reply.started":"2026-01-22T12:47:57.424596Z","shell.execute_reply":"2026-01-22T12:47:58.211321Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"solution_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:47:58.212670Z","iopub.execute_input":"2026-01-22T12:47:58.212952Z","iopub.status.idle":"2026-01-22T12:47:58.225811Z","shell.execute_reply.started":"2026-01-22T12:47:58.212928Z","shell.execute_reply":"2026-01-22T12:47:58.225100Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for img_type in img_type_list:\n    submission_df_train = pd.read_csv(f'train_submission_{img_type}.csv')\n    snr_score = score(solution_df, submission_df_train, row_id_column_name='id')\n    print(f\"Competition Score of {img_type}:\", snr_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:47:58.227383Z","iopub.execute_input":"2026-01-22T12:47:58.227592Z","iopub.status.idle":"2026-01-22T12:48:00.310981Z","shell.execute_reply.started":"2026-01-22T12:47:58.227578Z","shell.execute_reply":"2026-01-22T12:48:00.310323Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"solution_df['target_id'] = solution_df['id'].astype(str).apply(lambda x: x.split('_')[0])\n\nfor img_type in img_type_list:\n    submission_df_train = pd.read_csv(f'train_submission_{img_type}.csv')\n    total_snr = score(solution_df, submission_df_train, row_id_column_name='id')\n    print(f\"Total Competition Score of {img_type}: {total_snr:.4f}\")\n    print(\"-\" * 30)\n\n    # ---- 個別スコアの計算ロジック ----\n    submission_df_train['target_id'] = submission_df_train['id'].astype(str).apply(lambda x: x.split('_')[0])\n    target_ids = submission_df_train['target_id'].unique()\n    \n    id_scores = []\n    for tid in target_ids:\n        # IDごとの部分データフレームを抽出\n        sub_subset = submission_df_train[submission_df_train['target_id'] == tid]\n        sol_subset = solution_df[solution_df['target_id'] == tid]\n            \n        s = score(sol_subset, sub_subset, row_id_column_name='id')\n        id_scores.append({'target_id': tid, 'score': s})\n\n    result_df = pd.DataFrame(id_scores)\n    result_df = result_df.sort_values(by='score', ascending=False)\n    \n    # --- Top 5 (Best) ---\n    print(f\"Top 5 Scores for {img_type} (Best):\")\n    print(result_df.head(5).to_string(index=False))\n    \n    # --- Worst 5 ---\n    print(f\"\\nWorst 5 Scores for {img_type}:\")\n    print(result_df.tail(5).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-22T12:48:00.311636Z","iopub.execute_input":"2026-01-22T12:48:00.311841Z","iopub.status.idle":"2026-01-22T12:48:05.055029Z","shell.execute_reply.started":"2026-01-22T12:48:00.311825Z","shell.execute_reply":"2026-01-22T12:48:05.054358Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"`0006`<br>\nバイラテラルさらに強化版s1_pp: `18.7418`<br>\nバイラテラル強化版s1_pp必選択: `18.7608`<br>\n以前のもの: `16.8922`<br>\n\nバイラテラルさらに強化版の内訳：<br>\n\nTop 5 Scores for 0006 (Best):\ntarget_id     score\n 31175200 20.630814\n 10140238 20.282176\n 36494663 20.163257\n  7663343 18.896210\n 19030958 18.893873\n\nWorst 5 Scores for 0006:\ntarget_id     score\n 19585145 18.777351\n 31294838 18.362224\n 35187032 17.993881\n 11842146 15.524151\n 32650710 13.889869\n\n\n \n以前のものの内訳：<br>\n\nTotal Competition Score of 0006: 16.8922\n\nTop 5 Scores for 0006 (Best):\ntarget_id     score\n  7663343 20.695714\n 10140238 18.253379\n 31294838 17.946244\n 19030958 17.638679\n 19585145 16.523120\n\nWorst 5 Scores for 0006:\ntarget_id     score\n 31175200 16.362380\n 11842146 15.731139\n 32650710 13.623466\n 36494663 12.509709\n 35187032 12.503815","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}