{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ntrain_df = pd.read_csv(\"/kaggle/input/physionet-ecg-image-digitization/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/physionet-ecg-image-digitization/test.csv\")\n\nprint(\"Train shape:\", train_df.shape)\nprint(\"Test shape:\", test_df.shape)\ntrain_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T15:31:07.169886Z","iopub.execute_input":"2025-10-24T15:31:07.170104Z","iopub.status.idle":"2025-10-24T15:31:07.667671Z","shell.execute_reply.started":"2025-10-24T15:31:07.170083Z","shell.execute_reply":"2025-10-24T15:31:07.666665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport os\n\nsample_id = \"735384893\"\nsample_path = f\"/kaggle/input/physionet-ecg-image-digitization/train/{sample_id}\"\n\n# Get the first image file\nimg_files = sorted([f for f in os.listdir(sample_path) if f.endswith(\".png\")])\nprint(\"Found image files:\", img_files)\n\n# Load the first image\nimg_path = os.path.join(sample_path, img_files[0])\nimg = cv2.imread(img_path)\n\nplt.figure(figsize=(10, 4))\nplt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\nplt.title(f\"ECG Image - {img_files[0]}\")\nplt.axis(\"off\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T15:31:07.669452Z","iopub.execute_input":"2025-10-24T15:31:07.669724Z","iopub.status.idle":"2025-10-24T15:31:08.576039Z","shell.execute_reply.started":"2025-10-24T15:31:07.669702Z","shell.execute_reply":"2025-10-24T15:31:08.574781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\n\n# --- 0. IMAGE LOADING ---\n# NOTE: Ensure this path correctly loads the image you are currently testing.\nsample_id = \"735384893\"\nsample_path = f\"/kaggle/input/physionet-ecg-image-digitization/train/{sample_id}\"\nimg_files = sorted([f for f in os.listdir(sample_path) if f.endswith(\".png\")])\n\nif img_files:\n    img_path = os.path.join(sample_path, img_files[0])\n    img = cv2.imread(img_path)\nelse:\n    print(\"Error: No image found in the specified directory. Please check the path.\")\n    img = None\n\nif img is not None:\n    # =========================================================================\n    # === 1. IMAGE PREPROCESSING (Aggressive Cleaning) ===\n    # =========================================================================\n\n    # Convert to grayscale and invert colors\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    inverted = cv2.bitwise_not(gray)\n\n    # Apply binary threshold (Fixed at 150, as per your previous code)\n    _, thresh = cv2.threshold(inverted, 150, 255, cv2.THRESH_BINARY)\n    \n    # 1. Morphological OPENING: Removes small noise specks\n    kernel_open = np.ones((2, 2), np.uint8)\n    cleaned = cv2.morphologyEx(thresh, cv2.MORPH_OPEN, kernel_open)\n\n    # 2. CRITICAL FIX: AGGRESSIVE MORPHOLOGICAL CLOSING: Bridges gaps in the waveform\n    # Kernel (9, 1) is used to aggressively connect broken segments horizontally.\n    kernel_close = np.ones((9, 1), np.uint8) \n    cleaned = cv2.morphologyEx(cleaned, cv2.MORPH_CLOSE, kernel_close)\n\n    # =========================================================================\n    # === 2. VERIFICATION PLOT ===\n    # =========================================================================\n    plt.figure(figsize=(10, 6))\n    plt.imshow(cleaned, cmap='gray')\n    plt.title('Final Cleaned ECG Image (Ready for Tracing)')\n    plt.show()\n\n    print(\"\\nImage cleaning complete. The variable 'cleaned' now holds the binary image.\")\nelse:\n    # Define cleaned as None so the next code block doesn't crash\n    cleaned = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T15:31:08.577433Z","iopub.execute_input":"2025-10-24T15:31:08.577739Z","iopub.status.idle":"2025-10-24T15:31:09.184867Z","shell.execute_reply.started":"2025-10-24T15:31:08.577714Z","shell.execute_reply":"2025-10-24T15:31:09.183840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport re\nfrom scipy.signal import butter, filtfilt, detrend\nfrom tqdm import tqdm\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\n\n# === PATHS ===\ntrain_path = \"/kaggle/input/physionet-ecg-image-digitization/train\"\ntest_path = \"/kaggle/input/physionet-ecg-image-digitization/test\"\nrefined_path = \"/kaggle/working/final_digitized_signals\"\nos.makedirs(refined_path, exist_ok=True)\noutput_path = \"/kaggle/working/submission.csv\"\n\n# === STANDARD LEADS ===\nLEAD_NAMES = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF',\n              'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\n\n# === CPU FILTER HELPERS ===\ndef butter_lowpass_filter(data, cutoff=40, fs=500, order=4):\n    b, a = butter(order, cutoff / (0.5 * fs), btype='low')\n    y = filtfilt(b, a, data)\n    return y\n\ndef refine_signal(signal):\n    signal = np.array(signal, dtype=np.float32)\n    signal = np.nan_to_num(signal)\n    signal = detrend(signal)\n    signal = butter_lowpass_filter(signal)\n    signal = (signal - np.min(signal)) / (np.max(signal) - np.min(signal) + 1e-9)\n    return signal\n\n# === IMAGE TO SIGNAL DIGITIZATION ===\ndef extract_signals_from_image(image_path):\n    image = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    if image is None:\n        raise FileNotFoundError(image_path)\n    image = cv2.bitwise_not(image)\n    image = cv2.normalize(image, None, 0, 255, cv2.NORM_MINMAX)\n    edges = cv2.Canny(image, 50, 150)\n\n    h, w = edges.shape\n    lead_h = h // 12\n    leads = []\n    for i in range(12):\n        y1, y2 = i * lead_h, (i + 1) * lead_h\n        lead_img = edges[y1:y2, :]\n        signal = np.mean(lead_img, axis=0)\n        signal = refine_signal(signal)\n        leads.append(signal)\n    return np.array(leads)\n\n# === DIGITIZE ALL TRAIN IMAGES ===\nprint(\"⚙️ Digitizing and refining all ECG images (CPU)...\\n\")\nfor folder in tqdm(os.listdir(train_path)):\n    folder_path = os.path.join(train_path, folder)\n    if not os.path.isdir(folder_path):\n        continue\n\n    png_files = [f for f in os.listdir(folder_path) if f.endswith(\".png\")]\n    if not png_files:\n        continue\n\n    img_path = os.path.join(folder_path, png_files[0])\n    out_csv = os.path.join(refined_path, f\"{folder}.csv\")\n    if os.path.exists(out_csv):\n        continue\n\n    try:\n        signals = extract_signals_from_image(img_path)\n        df = pd.DataFrame(signals.T, columns=LEAD_NAMES)\n        df.to_csv(out_csv, index=False)\n    except Exception as e:\n        print(f\"❌ Error {folder}: {e}\")\n\nprint(\"\\n✅ All ECG signals refined and digitized (CPU used)!\")\nprint(f\"📂 Output: {refined_path}\")\n\n# === RMS Helper ===\ndef rms(signal):\n    signal = np.array(signal, dtype=np.float32)\n    return np.sqrt(np.mean(np.square(signal)))\n\n# === CREATE SUBMISSION ===\nprint(\"\\n🧠 Generating submission file...\")\ntest_ids = []\nfor root, _, files in os.walk(test_path):\n    for file in files:\n        if file.endswith(\".png\"):\n            test_ids.append(os.path.splitext(file)[0])\ntest_ids = sorted(list(set(test_ids)))\n\nsubmission_rows = []\nfor file_id in tqdm(test_ids):\n    num_match = re.findall(r\"\\d+\", file_id)\n    matched_csv = None\n    for csv_file in os.listdir(refined_path):\n        if any(num in csv_file for num in num_match):\n            matched_csv = os.path.join(refined_path, csv_file)\n            break\n\n    if matched_csv and os.path.exists(matched_csv):\n        df = pd.read_csv(matched_csv)\n        for i, lead in enumerate(LEAD_NAMES):\n            value = rms(df[lead])\n            submission_rows.append({\"id\": f\"{file_id}_{i}_{lead}\", \"value\": value})\n    else:\n        for i, lead in enumerate(LEAD_NAMES):\n            submission_rows.append({\"id\": f\"{file_id}_{i}_{lead}\", \"value\": 0.0})\n\nsubmission = pd.DataFrame(submission_rows)\nsubmission.to_csv(output_path, index=False)\n\nprint(\"\\n✅ submission.csv created successfully in full Kaggle format!\")\nprint(f\"📂 Saved to: {output_path}\")\nprint(submission.head(12))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-25T19:39:42.962141Z","iopub.execute_input":"2025-10-25T19:39:42.962453Z","iopub.status.idle":"2025-10-25T19:47:12.637269Z","shell.execute_reply.started":"2025-10-25T19:39:42.962428Z","shell.execute_reply":"2025-10-25T19:47:12.635722Z"}},"outputs":[],"execution_count":null}]}