{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nimport random\nimport os\nfrom PIL import Image\n\n# --- 1. CONFIGURATION AND UTILITIES ---\n\nDATA_DIR = Path('/kaggle/input/physionet-ecg-image-digitization/train') \nTRAIN_META = Path('/kaggle/input/physionet-ecg-image-digitization/train.csv')\n\nLEADS = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\nIMAGE_VARIANTS = {\n    '0001': 'Original Color Image (Clean)',\n    '0003': 'Printed & Scanned (Color)',\n    '0004': 'Printed & Scanned (B/W)',\n    '0005': 'Mobile Photo (Color Print)',\n    '0006': 'Mobile Photo (Screen)',\n    '0009': 'Mobile Photo (Stained/Soaked)',\n    '0010': 'Mobile Photo (Extensive Damage)',\n    '0011': 'Scan with Mold (Color)',\n    '0012': 'Scan with Mold (B/W)',\n}\n\ndef load_signal(id_: int) -> pd.DataFrame:\n    \"\"\"Loads the ground-truth time-series signal CSV for a given ID.\"\"\"\n    # Crucially, convert id_ to string to avoid the TypeError in path concatenation\n    id_str = str(id_)\n    signal_path = DATA_DIR / id_str / f\"{id_str}.csv\"\n    \n    if not signal_path.exists():\n        print(f\"Warning: Signal file not found at {signal_path}\")\n        return pd.DataFrame()\n        \n    return pd.read_csv(signal_path)\n\ndef load_image(id_: int, suffix: str) -> Image.Image | None:\n    \"\"\"Loads a specific ECG image variant for a given ID and suffix.\"\"\"\n    id_str = str(id_)\n    \n    # Try the specific file name pattern\n    img_path = DATA_DIR / id_str / f\"{id_str}-{suffix}.png\"\n    \n    if not img_path.exists():\n        # Fallback: find any PNG in the folder that matches the suffix pattern\n        # This is a robust check, but the competition uses the exact naming.\n        img_candidates = list((DATA_DIR / id_str).glob(f\"*-{suffix}.png\"))\n        if img_candidates:\n            img_path = img_candidates[0]\n        else:\n            return None\n            \n    try:\n        return Image.open(img_path)\n    except Exception as e:\n        print(f\"Error loading image {img_path}: {e}\")\n        return None\n\ndef plot_ecg_signal(signal_df: pd.DataFrame, id_: int, fs: int):\n    \"\"\"Plots the 12-lead ECG time series from the ground truth.\"\"\"\n    if signal_df.empty:\n        print(\"Empty signal data, skipping plot.\")\n        return\n\n    # Calculate time vector\n    time = np.arange(len(signal_df)) / fs\n    \n    # Define a 4x3 grid for the 12 leads\n    fig, axes = plt.subplots(4, 3, figsize=(18, 12), sharex=False, sharey=False)\n    axes = axes.flatten()\n    \n    # Titles for the subplot (based on standard clinical arrangement)\n    lead_layout = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\n    \n    for i, lead in enumerate(lead_layout):\n        ax = axes[i]\n        \n        # Check if the lead exists in the DataFrame (should for ground truth)\n        if lead in signal_df.columns:\n            # Check length of the current lead\n            lead_len_sec = len(signal_df[lead]) / fs\n            \n            # Recalculate time for the current lead if it's not lead II (10s)\n            current_time = np.arange(len(signal_df[lead])) / fs\n            \n            ax.plot(current_time, signal_df[lead], color='#1f77b4', linewidth=1)\n            ax.set_title(f'Lead {lead} ({lead_len_sec:.1f}s, {len(signal_df[lead])} samples)', fontsize=10)\n            ax.set_ylabel('Amplitude (mV)', fontsize=8)\n            ax.set_xlabel('Time (s)', fontsize=8)\n            ax.grid(color='gray', linestyle=':', linewidth=0.5, alpha=0.5)\n            \n    fig.suptitle(f'Ground Truth 12-Lead ECG Signal (ID: {id_}, FS: {fs} Hz)', fontsize=16, y=1.02)\n    plt.tight_layout(rect=[0, 0, 1, 0.98])\n    plt.show()\n\n# --- 2. MAIN EDA EXECUTION ---\n\ndef run_eda(sample_size_meta=5, sample_size_signal=5):\n    \"\"\"Executes the full EDA pipeline.\"\"\"\n    sns.set_theme(style=\"whitegrid\")\n    \n    # --- Step 1: Metadata Analysis (train.csv) ---\n    print(\"=\" * 60)\n    print(\"STEP 1: METADATA ANALYSIS (train.csv)\")\n    print(\"=\" * 60)\n    \n    try:\n        # Check if the train metadata file exists\n        if not TRAIN_META.exists():\n            print(f\"Error: Metadata file not found at {TRAIN_META}. Cannot proceed.\")\n            print(\"Please ensure 'train.csv' is in the current directory or adjust the TRAIN_META path.\")\n            return\n\n        train_meta = pd.read_csv(TRAIN_META)\n        print(f\"Loaded {len(train_meta)} records.\")\n        print(\"\\n--- Basic Info ---\")\n        print(train_meta.info())\n        print(\"\\n--- Descriptive Statistics ---\")\n        print(train_meta.describe())\n        \n        # A. Sampling Frequency (fs) Distribution\n        plt.figure(figsize=(8, 5))\n        sns.histplot(train_meta['fs'], bins=10, kde=True, color='purple')\n        plt.title('Distribution of Sampling Frequencies (FS)')\n        plt.xlabel('Sampling Frequency (Hz)')\n        plt.ylabel('Count')\n        plt.show()\n        \n        # B. Signal Length Verification\n        train_meta['expected_len'] = train_meta['fs'] * 10\n        print(\"\\n--- Signal Length Verification ---\")\n        verification_match = (train_meta['sig_len'] == train_meta['expected_len']).all()\n        print(f\"All sig_len match 10s * fs: {verification_match}\")\n        \n        # C. Unique IDs\n        if train_meta['id'].duplicated().any():\n            print(f\"Warning: Found {train_meta['id'].duplicated().sum()} duplicate IDs.\")\n        else:\n            print(\"No duplicate IDs found.\")\n            \n    except Exception as e:\n        print(f\"An error occurred during metadata analysis: {e}\")\n        return\n        \n    # --- Step 2: Ground Truth Signal Analysis (Sampling) ---\n    print(\"\\n\\n\" + \"=\" * 60)\n    print(\"STEP 2: GROUND TRUTH SIGNAL ANALYSIS\")\n    print(\"=\" * 60)\n\n    # Filter for IDs that have directories (to avoid errors on partial downloads)\n    available_ids = [int(p.name) for p in DATA_DIR.iterdir() if p.is_dir() and p.name.isdigit()]\n    \n    if not available_ids:\n        print(f\"Error: No data directories found in {DATA_DIR}. Cannot perform signal/image analysis.\")\n        return\n\n    # Sample a few IDs for plotting\n    sample_ids = random.sample(available_ids, min(sample_size_signal, len(available_ids)))\n    \n    all_leads_data = []\n\n    for id_ in sample_ids:\n        meta_row = train_meta[train_meta['id'] == id_].iloc[0]\n        fs = meta_row['fs']\n        \n        signal_df = load_signal(id_)\n        \n        if signal_df.empty:\n            continue\n            \n        print(f\"\\nProcessing ID {id_}: FS={fs}, Length={len(signal_df)} samples.\")\n        \n        # A. Plot the signal\n        plot_ecg_signal(signal_df, id_, fs)\n        \n        # B. Lead-specific length check (Important Insight)\n        print(f\"Lead lengths at FS={fs}:\")\n        for lead in LEADS:\n            if lead in signal_df.columns:\n                length = len(signal_df[lead])\n                duration = length / fs\n                expected_duration = 10.0 if lead == 'II' else 2.5\n                print(f\"  {lead}: {length} samples ({duration:.1f}s). Expected: {expected_duration:.1f}s.\")\n                \n                # For combined amplitude analysis\n                temp_df = signal_df[[lead]].copy()\n                temp_df['Lead'] = lead\n                temp_df['Value'] = temp_df[lead]\n                all_leads_data.append(temp_df[['Lead', 'Value']])\n                \n    # C. Combined Amplitude Distribution\n    if all_leads_data:\n        combined_df = pd.concat(all_leads_data)\n        plt.figure(figsize=(12, 6))\n        sns.boxplot(x='Lead', y='Value', data=combined_df, palette='Spectral')\n        plt.title('Amplitude (mV) Distribution Across 12 Leads (Sampled)')\n        plt.ylabel('Amplitude (mV)')\n        plt.xlabel('ECG Lead')\n        plt.show()\n\n\n    # --- Step 3: Image Degradation Comparison ---\n    print(\"\\n\\n\" + \"=\" * 60)\n    print(\"STEP 3: IMAGE DEGRADATION ANALYSIS\")\n    print(\"=\" * 60)\n\n    # Use the first sampled ID for consistency\n    if sample_ids:\n        id_ = sample_ids[0]\n        print(f\"Visualizing all image variants for sample ID {id_}\")\n        \n        fig, axes = plt.subplots(4, 3, figsize=(15, 20))\n        axes = axes.flatten()\n        \n        for i, (suffix, title) in enumerate(IMAGE_VARIANTS.items()):\n            if i >= len(axes): break # safety break\n            \n            img = load_image(id_, suffix)\n            ax = axes[i]\n            \n            if img:\n                ax.imshow(img, aspect='auto')\n                ax.set_title(f'({suffix}) {title}', fontsize=10)\n                ax.axis('off')\n            else:\n                ax.set_title(f'({suffix}) Image Missing', fontsize=10)\n                ax.axis('off')\n\n        # Hide unused subplots\n        for j in range(len(IMAGE_VARIANTS), len(axes)):\n            fig.delaxes(axes[j])\n            \n        fig.suptitle(f'Image Degradation Variants (ID: {id_})', fontsize=16, y=1.01)\n        plt.tight_layout(rect=[0, 0, 1, 0.98])\n        plt.show()\n        \n        print(\"\\nKey Visual Observations:\")\n        print(\" - Observe grid line visibility, color degradation, rotation, and artifacts.\")\n        print(\" - The original image (0001) is the cleanest reference for the signal location.\")\n        print(\" - Types like 0005 (photo) often have perspective distortion and poor contrast.\")\n        print(\" - Types 0009/0010/0011/0012 show severe artifacts (stains, mold, damage).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-22T14:27:50.294122Z","iopub.execute_input":"2025-10-22T14:27:50.294447Z","iopub.status.idle":"2025-10-22T14:27:51.557603Z","shell.execute_reply.started":"2025-10-22T14:27:50.294421Z","shell.execute_reply":"2025-10-22T14:27:51.556797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"run_eda()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-22T14:28:34.668059Z","iopub.execute_input":"2025-10-22T14:28:34.668501Z","iopub.status.idle":"2025-10-22T14:29:08.518376Z","shell.execute_reply.started":"2025-10-22T14:28:34.668475Z","shell.execute_reply":"2025-10-22T14:29:08.517366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}