{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PhysioNet ECG Digitization - Initial EDA","metadata":{"_uuid":"82ae83bf-f90b-4a6a-8175-aabfec79a8e7","_cell_guid":"d99ddf1b-c599-45c6-a94b-1b1fdd44439a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"markdown","source":"## 1. Setup and Imports","metadata":{"_uuid":"50693dac-2d02-4509-aa1d-c4c242178497","_cell_guid":"47b222c1-1463-4e8b-9c6b-ecf07e1df115","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nfrom PIL import Image\nimport warnings\n\nwarnings.filterwarnings('ignore')\nsns.set_style('whitegrid')\nplt.rcParams['figure.figsize'] = (15, 6)\n\n# Set paths for Kaggle environment\nDATA_PATH = Path('/kaggle/input/physionet-ecg-image-digitization')\nTRAIN_PATH = DATA_PATH / 'train'\nTEST_PATH = DATA_PATH / 'test'\n\nprint(\"Setup complete!\")","metadata":{"_uuid":"c716a4d9-e9a3-4a96-82ad-20a2faf8e5c0","_cell_guid":"a3d2d1d1-63f0-4774-b79b-1f2c9fd3a492","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:24.057650Z","iopub.execute_input":"2025-10-21T18:49:24.058491Z","iopub.status.idle":"2025-10-21T18:49:24.066219Z","shell.execute_reply.started":"2025-10-21T18:49:24.058460Z","shell.execute_reply":"2025-10-21T18:49:24.065048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Load Metadata","metadata":{"_uuid":"37548f3a-a43d-49bc-a11c-b5697a72ec1f","_cell_guid":"2cefeadd-4c88-4940-8b33-d5f3bbf1eb02","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Load training and test metadata\ntrain_df = pd.read_csv(DATA_PATH / 'train.csv')\ntest_df = pd.read_csv(DATA_PATH / 'test.csv')\nsample_submission = pd.read_parquet(DATA_PATH / 'sample_submission.parquet')\n\nprint(f\"Training samples: {len(train_df)}\")\nprint(f\"Test samples: {len(test_df)}\")\nprint(f\"Submission rows: {len(sample_submission)}\")","metadata":{"_uuid":"2a1c6347-9c3a-4f4c-bfb2-016fe2afda6c","_cell_guid":"98d9ad96-6e51-4473-a5ec-3fcbc81e1e18","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:24.068033Z","iopub.execute_input":"2025-10-21T18:49:24.068344Z","iopub.status.idle":"2025-10-21T18:49:24.360511Z","shell.execute_reply.started":"2025-10-21T18:49:24.068307Z","shell.execute_reply":"2025-10-21T18:49:24.359540Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Training Data Exploration","metadata":{"_uuid":"db25eea4-2fa3-4909-bdeb-0fd7c2ceb6d2","_cell_guid":"c48c1c79-626c-40ad-bf2b-cd1038ea3cc5","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Display basic info about training data\nprint(\"Training Data Info:\")\nprint(train_df.head(10))\nprint(\"\\nData Types:\")\nprint(train_df.dtypes)\nprint(\"\\nBasic Statistics:\")\nprint(train_df.describe())","metadata":{"_uuid":"693ee048-f6c8-48b8-b923-636319aa5d93","_cell_guid":"9f84b89b-72e7-4b1d-a35b-138e724b78f6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:24.361420Z","iopub.execute_input":"2025-10-21T18:49:24.361731Z","iopub.status.idle":"2025-10-21T18:49:24.405106Z","shell.execute_reply.started":"2025-10-21T18:49:24.361700Z","shell.execute_reply":"2025-10-21T18:49:24.404245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nprint(\"Missing values in training data:\")\nprint(train_df.isnull().sum())","metadata":{"_uuid":"870cee61-53ad-41bf-a1e1-8c780dcadf34","_cell_guid":"a1d01449-5df5-4b0a-b189-0d2ddacdcd9a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:24.405965Z","iopub.execute_input":"2025-10-21T18:49:24.406219Z","iopub.status.idle":"2025-10-21T18:49:24.412805Z","shell.execute_reply.started":"2025-10-21T18:49:24.406200Z","shell.execute_reply":"2025-10-21T18:49:24.411798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze sampling frequencies\nprint(\"Sampling Frequency Distribution:\")\nprint(train_df['fs'].value_counts().sort_index())\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\ntrain_df['fs'].value_counts().sort_index().plot(kind='bar')\nplt.title('Sampling Frequency Distribution')\nplt.xlabel('Sampling Frequency (Hz)')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\n\nplt.subplot(1, 2, 2)\ntrain_df['sig_len'].value_counts().sort_index().plot(kind='bar')\nplt.title('Signal Length Distribution')\nplt.xlabel('Signal Length (samples)')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"fa850ec0-25b4-4374-8436-7b23dc2f4382","_cell_guid":"2d6d77c0-ffab-44ec-8d1f-9ac60dd3f432","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:24.415133Z","iopub.execute_input":"2025-10-21T18:49:24.415419Z","iopub.status.idle":"2025-10-21T18:49:25.077838Z","shell.execute_reply.started":"2025-10-21T18:49:24.415398Z","shell.execute_reply":"2025-10-21T18:49:25.076759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify the relationship: sig_len = 10 seconds * fs\ntrain_df['expected_sig_len'] = train_df['fs'] * 10\ntrain_df['sig_len_match'] = train_df['sig_len'] == train_df['expected_sig_len']\nprint(f\"All signal lengths match 10 seconds * fs: {train_df['sig_len_match'].all()}\")\nprint(f\"Percentage matching: {train_df['sig_len_match'].mean() * 100:.2f}%\")","metadata":{"_uuid":"63c82d7b-5f06-40bf-ba76-8a80e22d1f77","_cell_guid":"e007ba1a-e620-47e7-829f-5f3abdd4dfcf","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:25.078663Z","iopub.execute_input":"2025-10-21T18:49:25.079012Z","iopub.status.idle":"2025-10-21T18:49:25.087685Z","shell.execute_reply.started":"2025-10-21T18:49:25.078985Z","shell.execute_reply":"2025-10-21T18:49:25.086699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Test Data Exploration","metadata":{"_uuid":"0e4534ff-7cbc-4fcd-b911-ecabcefafb0f","_cell_guid":"c411b4ec-1a98-4ed6-bb55-34051d248fbd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Display basic info about test data\nprint(\"Test Data Info:\")\nprint(test_df.head(10))\nprint(\"\\nData Types:\")\nprint(test_df.dtypes)\nprint(\"\\nBasic Statistics:\")\nprint(test_df.describe())","metadata":{"_uuid":"913dbaa2-9f9e-4990-8177-361a31fa7c46","_cell_guid":"c22d0fc1-a472-45a0-acca-8a8cbafb1ae8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:25.088668Z","iopub.execute_input":"2025-10-21T18:49:25.089032Z","iopub.status.idle":"2025-10-21T18:49:25.122602Z","shell.execute_reply.started":"2025-10-21T18:49:25.089004Z","shell.execute_reply":"2025-10-21T18:49:25.121566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check lead distribution in test data\nprint(\"Lead Distribution in Test Data:\")\nprint(test_df['lead'].value_counts())\n\nplt.figure(figsize=(10, 5))\ntest_df['lead'].value_counts().plot(kind='bar')\nplt.title('ECG Lead Distribution in Test Set')\nplt.xlabel('Lead')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"cdb28c21-a38e-4099-a4f6-9ae6429b1dd3","_cell_guid":"62742341-1bd1-428f-ac6c-b11dc98a4ede","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:25.123688Z","iopub.execute_input":"2025-10-21T18:49:25.123990Z","iopub.status.idle":"2025-10-21T18:49:25.467674Z","shell.execute_reply.started":"2025-10-21T18:49:25.123967Z","shell.execute_reply":"2025-10-21T18:49:25.466748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze number_of_rows in test data\nprint(\"\\nNumber of rows distribution:\")\nprint(test_df['number_of_rows'].value_counts().sort_index())\n\n# Check if number_of_rows varies by lead\nprint(\"\\nNumber of rows by lead:\")\nprint(test_df.groupby('lead')['number_of_rows'].describe())","metadata":{"_uuid":"36d37fb6-6675-4f26-87dd-dd9b8e5013ae","_cell_guid":"dcdfb04e-274f-49a2-bd63-0fe5088b9cac","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:25.468671Z","iopub.execute_input":"2025-10-21T18:49:25.468973Z","iopub.status.idle":"2025-10-21T18:49:25.519850Z","shell.execute_reply.started":"2025-10-21T18:49:25.468943Z","shell.execute_reply":"2025-10-21T18:49:25.518939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. ECG Image Exploration","metadata":{"_uuid":"7cbe1e4a-493b-4587-b5b8-833e97d13ae9","_cell_guid":"e4f2dbfb-70f8-42f5-994b-4e4abee52089","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Get a sample ID to explore different image types\nsample_id = str(train_df['id'].iloc[0])\nprint(f\"Exploring images for sample ID: {sample_id}\")\n\n# Image segments based on data description\nimage_types = {\n    '0001': 'Original color ECG image',\n    '0003': 'Printed in color, scanned in color',\n    '0004': 'Printed in color, scanned in B&W',\n    '0005': 'Mobile photos of color printed images',\n    '0006': 'Mobile photos of ECGs on laptop screen',\n    '0009': 'Mobile photos of stained/soaked printed ECGs',\n    '0010': 'Mobile photos with extensive damage',\n    '0011': 'Scans of printed ECG with mold (color)',\n    '0012': 'Scans of printed ECG with mold (B&W)'\n}\n\n# Check which image files exist for this sample\nsample_dir = TRAIN_PATH / sample_id\nexisting_images = list(sample_dir.glob(f\"{sample_id}-*.png\"))\nprint(f\"\\nFound {len(existing_images)} images for this sample\")\nfor img_path in sorted(existing_images):\n    print(f\"  - {img_path.name}\")","metadata":{"_uuid":"a162e956-71f6-43a4-8c38-ae25d3919bd8","_cell_guid":"c80433ac-2040-4d6c-9ad0-78cda5d2e17f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:25.520773Z","iopub.execute_input":"2025-10-21T18:49:25.521077Z","iopub.status.idle":"2025-10-21T18:49:25.529588Z","shell.execute_reply.started":"2025-10-21T18:49:25.521056Z","shell.execute_reply":"2025-10-21T18:49:25.528585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize different image types for the same ECG\nfig, axes = plt.subplots(3, 3, figsize=(18, 12))\naxes = axes.flatten()\n\nfor idx, (segment, description) in enumerate(image_types.items()):\n    img_path = sample_dir / f\"{sample_id}-{segment}.png\"\n    if img_path.exists():\n        img = Image.open(img_path)\n        axes[idx].imshow(img)\n        axes[idx].set_title(f\"{segment}: {description}\", fontsize=9)\n        axes[idx].axis('off')\n        # Print image dimensions\n        print(f\"{segment}: {img.size} (W x H), Mode: {img.mode}\")\n    else:\n        axes[idx].text(0.5, 0.5, 'Image not found', ha='center', va='center')\n        axes[idx].set_title(f\"{segment}: Not available\")\n        axes[idx].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"2657b2ca-9c85-4f85-b7ca-dee391d3a1f2","_cell_guid":"d7e6ebbf-b470-4567-b080-4dae12f77f37","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:25.530603Z","iopub.execute_input":"2025-10-21T18:49:25.530911Z","iopub.status.idle":"2025-10-21T18:49:39.540117Z","shell.execute_reply.started":"2025-10-21T18:49:25.530850Z","shell.execute_reply":"2025-10-21T18:49:39.538960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize test image\ntest_id = str(test_df['id'].iloc[0])\nprint(f\"Sample test image ID: {test_id}\")\n\ntest_img_path = TEST_PATH / f\"{test_id}.png\"\nif test_img_path.exists():\n    test_img = Image.open(test_img_path)\n    print(f\"Test image dimensions: {test_img.size} (W x H), Mode: {test_img.mode}\")\n\n    plt.figure(figsize=(16, 8))\n    plt.imshow(test_img)\n    plt.title(f\"Test Image: {test_id}\")\n    plt.axis('off')\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"Test image not found\")","metadata":{"_uuid":"88f30f44-8642-4a8b-9f53-2118746d6bac","_cell_guid":"6566cd58-c459-4b2b-9ba8-a90d2b395b46","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:39.541581Z","iopub.execute_input":"2025-10-21T18:49:39.541919Z","iopub.status.idle":"2025-10-21T18:49:40.445165Z","shell.execute_reply.started":"2025-10-21T18:49:39.541890Z","shell.execute_reply":"2025-10-21T18:49:40.444093Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Time Series Data Exploration","metadata":{"_uuid":"c1bba4c4-6cf9-42b6-bf69-20477761fe89","_cell_guid":"4dba59c3-6ec5-4a29-8bfa-d79d72413262","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Load time series data for a sample\nsample_csv_path = sample_dir / f\"{sample_id}.csv\"\necg_data = pd.read_csv(sample_csv_path)\n\nprint(\"ECG Time Series Data Shape:\", ecg_data.shape)\nprint(\"\\nColumns (ECG Leads):\", ecg_data.columns.tolist())\nprint(\"\\nFirst few rows:\")\nprint(ecg_data.head())\nprint(\"\\nBasic Statistics:\")\nprint(ecg_data.describe())","metadata":{"_uuid":"44a4d05a-6811-46f7-b45e-46ffffc33ce9","_cell_guid":"7dc771f7-7f03-4435-bd8d-61ed9ae3ec66","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:40.446215Z","iopub.execute_input":"2025-10-21T18:49:40.446496Z","iopub.status.idle":"2025-10-21T18:49:40.493865Z","shell.execute_reply.started":"2025-10-21T18:49:40.446474Z","shell.execute_reply":"2025-10-21T18:49:40.493022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify the expected 12 leads\nexpected_leads = ['I', 'II', 'III', 'aVR', 'aVL', 'aVF', 'V1', 'V2', 'V3', 'V4', 'V5', 'V6']\nprint(f\"Expected leads: {expected_leads}\")\nprint(f\"Actual leads: {ecg_data.columns.tolist()}\")\nprint(f\"All leads present: {set(expected_leads) == set(ecg_data.columns)}\")","metadata":{"_uuid":"cc8048d4-79c2-489e-974b-33f28fc014b2","_cell_guid":"50080cdc-5c30-4571-a3c3-0adaaf6df3dc","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:40.496723Z","iopub.execute_input":"2025-10-21T18:49:40.497012Z","iopub.status.idle":"2025-10-21T18:49:40.502857Z","shell.execute_reply.started":"2025-10-21T18:49:40.496991Z","shell.execute_reply":"2025-10-21T18:49:40.501846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize all 12 leads for the sample ECG\nfig, axes = plt.subplots(12, 1, figsize=(16, 20))\n\nfor idx, lead in enumerate(expected_leads):\n    axes[idx].plot(ecg_data[lead], linewidth=0.8)\n    axes[idx].set_ylabel(f'{lead} (mV)', fontsize=10)\n    axes[idx].grid(True, alpha=0.3)\n    axes[idx].set_xlim(0, len(ecg_data))\n\n    # Add some statistics\n    mean_val = ecg_data[lead].mean()\n    std_val = ecg_data[lead].std()\n    axes[idx].set_title(f'{lead} - Mean: {mean_val:.3f} mV, Std: {std_val:.3f} mV',\n                       fontsize=9, loc='right')\n\naxes[-1].set_xlabel('Sample Index', fontsize=10)\nfig.suptitle(f'12-Lead ECG Time Series for Sample {sample_id}', fontsize=14, y=0.995)\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"fd1f2368-0904-466a-817c-705f06114441","_cell_guid":"eaf7f423-f808-4673-b90c-c71b7c784cbc","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:40.503698Z","iopub.execute_input":"2025-10-21T18:49:40.503963Z","iopub.status.idle":"2025-10-21T18:49:43.298644Z","shell.execute_reply.started":"2025-10-21T18:49:40.503943Z","shell.execute_reply":"2025-10-21T18:49:43.297612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate time axis based on sampling frequency\nsample_fs = train_df[train_df['id'] == int(sample_id)]['fs'].values[0]\ntime_axis = np.arange(len(ecg_data)) / sample_fs\n\nprint(f\"Sampling frequency: {sample_fs} Hz\")\nprint(f\"Number of samples: {len(ecg_data)}\")\nprint(f\"Duration: {time_axis[-1]:.2f} seconds\")\n\n# Plot with time axis\nfig, axes = plt.subplots(4, 3, figsize=(18, 12))\naxes = axes.flatten()\n\nfor idx, lead in enumerate(expected_leads):\n    axes[idx].plot(time_axis, ecg_data[lead], linewidth=0.8, color='darkblue')\n    axes[idx].set_title(f'{lead}', fontsize=11, fontweight='bold')\n    axes[idx].set_ylabel('mV', fontsize=9)\n    axes[idx].grid(True, alpha=0.3)\n    axes[idx].set_xlim(0, 10)\n\nfor ax in axes[-3:]:\n    ax.set_xlabel('Time (seconds)', fontsize=9)\n\nfig.suptitle(f'12-Lead ECG with Time Axis (fs={sample_fs} Hz)', fontsize=14)\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"456564d8-3bcd-473d-8e1a-a19ae3c36c7b","_cell_guid":"dd4b1784-f1e0-40b4-9013-63346736970f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:43.299566Z","iopub.execute_input":"2025-10-21T18:49:43.299946Z","iopub.status.idle":"2025-10-21T18:49:46.233687Z","shell.execute_reply.started":"2025-10-21T18:49:43.299909Z","shell.execute_reply":"2025-10-21T18:49:46.232735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze signal amplitude ranges across all leads\namplitude_stats = pd.DataFrame({\n    'Lead': expected_leads,\n    'Min': [ecg_data[lead].min() for lead in expected_leads],\n    'Max': [ecg_data[lead].max() for lead in expected_leads],\n    'Mean': [ecg_data[lead].mean() for lead in expected_leads],\n    'Std': [ecg_data[lead].std() for lead in expected_leads],\n    'Range': [ecg_data[lead].max() - ecg_data[lead].min() for lead in expected_leads]\n})\n\nprint(\"Amplitude Statistics Across All Leads:\")\nprint(amplitude_stats.to_string(index=False))\n\n# Visualize amplitude ranges\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\naxes[0].bar(amplitude_stats['Lead'], amplitude_stats['Range'])\naxes[0].set_title('Signal Range (Max - Min) by Lead')\naxes[0].set_xlabel('Lead')\naxes[0].set_ylabel('Range (mV)')\naxes[0].tick_params(axis='x', rotation=45)\n\naxes[1].bar(amplitude_stats['Lead'], amplitude_stats['Std'])\naxes[1].set_title('Signal Standard Deviation by Lead')\naxes[1].set_xlabel('Lead')\naxes[1].set_ylabel('Std (mV)')\naxes[1].tick_params(axis='x', rotation=45)\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"c1e40507-3547-4c66-8907-563c798a3600","_cell_guid":"8a1545fb-5131-4cc2-a860-c143f407d841","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:46.234678Z","iopub.execute_input":"2025-10-21T18:49:46.234947Z","iopub.status.idle":"2025-10-21T18:49:46.827040Z","shell.execute_reply.started":"2025-10-21T18:49:46.234926Z","shell.execute_reply":"2025-10-21T18:49:46.826060Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Submission Format Analysis","metadata":{"_uuid":"70ce27d1-f0b7-4a48-b459-2b4dba724198","_cell_guid":"e57d0f8e-afa7-49b2-8bf9-9eb0b0b1b70d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Explore sample submission structure\nprint(\"Sample Submission Shape:\", sample_submission.shape)\nprint(\"\\nColumns:\", sample_submission.columns.tolist())\nprint(\"\\nFirst few rows:\")\nprint(sample_submission.head(20))\nprint(\"\\nLast few rows:\")\nprint(sample_submission.tail(20))","metadata":{"_uuid":"1139e992-e60b-404d-aeed-6d07fc8226b8","_cell_guid":"253d8193-50c5-4d5f-99ef-d264e3b91c39","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:46.828025Z","iopub.execute_input":"2025-10-21T18:49:46.828295Z","iopub.status.idle":"2025-10-21T18:49:46.838202Z","shell.execute_reply.started":"2025-10-21T18:49:46.828276Z","shell.execute_reply":"2025-10-21T18:49:46.837241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parse the composite ID format\nsample_submission['base_id_parsed'] = sample_submission['id'].str.split('_').str[0]\nsample_submission['row_id_parsed'] = sample_submission['id'].str.split('_').str[1].astype(int)\nsample_submission['lead_parsed'] = sample_submission['id'].str.split('_').str[2]\n\nprint(\"Parsed ID components:\")\nprint(sample_submission[['id', 'base_id_parsed', 'row_id_parsed', 'lead_parsed']].head(20))","metadata":{"_uuid":"02c8b4f1-5217-4e33-8172-2f5114b151a1","_cell_guid":"fd87323a-4695-4a45-8eba-76c622dede62","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:46.839124Z","iopub.execute_input":"2025-10-21T18:49:46.839366Z","iopub.status.idle":"2025-10-21T18:49:47.534688Z","shell.execute_reply.started":"2025-10-21T18:49:46.839347Z","shell.execute_reply":"2025-10-21T18:49:47.533684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify submission format matches test.csv structure\nprint(\"Unique base_ids in submission:\", sample_submission['base_id_parsed'].nunique())\nprint(\"Unique IDs in test.csv:\", test_df['id'].nunique())\n\n# Check if all test IDs are in submission\ntest_ids = set(test_df['id'].astype(str))\nsubmission_base_ids = set(sample_submission['base_id_parsed'])\nprint(f\"\\nAll test IDs in submission: {test_ids.issubset(submission_base_ids)}\")","metadata":{"_uuid":"6748a1f3-6234-46b8-aded-59a4adbc8df8","_cell_guid":"b7638ce6-7021-4397-955e-df008e0b87d9","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:47.535666Z","iopub.execute_input":"2025-10-21T18:49:47.535943Z","iopub.status.idle":"2025-10-21T18:49:47.564315Z","shell.execute_reply.started":"2025-10-21T18:49:47.535922Z","shell.execute_reply":"2025-10-21T18:49:47.563417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze rows per lead in submission\nprint(\"Rows per lead in submission:\")\nrows_per_lead = sample_submission.groupby('lead_parsed')['row_id_parsed'].agg(['count', 'min', 'max'])\nprint(rows_per_lead)\n\n# Check if Lead II has more rows (10 seconds vs 2.5 seconds)\nlead_ii_rows = sample_submission[sample_submission['lead_parsed'] == 'II']['row_id_parsed'].max() + 1\nother_lead_rows = sample_submission[sample_submission['lead_parsed'] == 'I']['row_id_parsed'].max() + 1\nprint(f\"\\nLead II rows: {lead_ii_rows}\")\nprint(f\"Other lead rows (example Lead I): {other_lead_rows}\")\nprint(f\"Ratio: {lead_ii_rows / other_lead_rows:.2f} (should be ~4 for 10s vs 2.5s)\")","metadata":{"_uuid":"626d26e4-677a-4609-b3d8-099dffd60459","_cell_guid":"177117c9-9129-4499-8b5d-860796b75d24","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:47.565433Z","iopub.execute_input":"2025-10-21T18:49:47.565697Z","iopub.status.idle":"2025-10-21T18:49:47.613216Z","shell.execute_reply.started":"2025-10-21T18:49:47.565675Z","shell.execute_reply":"2025-10-21T18:49:47.612404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For a single test sample, show expected submission structure\nsample_test_id = str(test_df['id'].iloc[0])\nsample_rows = sample_submission[sample_submission['base_id_parsed'] == sample_test_id]\n\nprint(f\"Submission rows for test ID {sample_test_id}:\")\nprint(f\"Total rows: {len(sample_rows)}\")\nprint(\"\\nRows by lead:\")\nprint(sample_rows.groupby('lead_parsed').size())\nprint(\"\\nSample rows:\")\nprint(sample_rows.head(30))","metadata":{"_uuid":"4831a176-9157-49ae-a1c6-04c3a7755185","_cell_guid":"e1b75fa8-489c-4e03-af25-8b57905f4f31","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:47.614081Z","iopub.execute_input":"2025-10-21T18:49:47.614323Z","iopub.status.idle":"2025-10-21T18:49:47.644067Z","shell.execute_reply.started":"2025-10-21T18:49:47.614305Z","shell.execute_reply":"2025-10-21T18:49:47.643033Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Image-to-Signal Comparison","metadata":{"_uuid":"0e87afa7-6cf1-49cb-9393-e4c14d71dffb","_cell_guid":"8ba8335a-14f2-4548-b27f-61ccf0e39acd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Side-by-side comparison of original image and reconstructed signal\nfig = plt.figure(figsize=(18, 12))\n\n# Show original image\ngs = fig.add_gridspec(13, 1, hspace=0.4)\nax_img = fig.add_subplot(gs[0, :])\nimg = Image.open(sample_dir / f\"{sample_id}-0001.png\")\nax_img.imshow(img)\nax_img.set_title(f'Original ECG Image (ID: {sample_id})', fontsize=12, fontweight='bold')\nax_img.axis('off')\n\n# Show time series for all 12 leads\nfor idx, lead in enumerate(expected_leads):\n    ax = fig.add_subplot(gs[idx+1, :])\n    ax.plot(time_axis, ecg_data[lead], linewidth=0.8, color='red')\n    ax.set_ylabel(lead, fontsize=9, rotation=0, labelpad=20)\n    ax.set_xlim(0, 10)\n    ax.grid(True, alpha=0.2)\n    ax.tick_params(labelsize=8)\n    if idx < 11:\n        ax.set_xticklabels([])\n    else:\n        ax.set_xlabel('Time (seconds)', fontsize=9)\n\nfig.suptitle('Image vs Ground Truth Time Series Comparison', fontsize=14, y=0.995)\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"0afb283a-40bc-4aa7-ac84-a4e7b2ccc6ba","_cell_guid":"248171d4-4014-414d-b414-b53a685391aa","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:47.644972Z","iopub.execute_input":"2025-10-21T18:49:47.645200Z","iopub.status.idle":"2025-10-21T18:49:49.441094Z","shell.execute_reply.started":"2025-10-21T18:49:47.645181Z","shell.execute_reply":"2025-10-21T18:49:49.440101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics for quick reference\nprint(\"=\" * 80)\nprint(\"DATASET SUMMARY\")\nprint(\"=\" * 80)\nprint(f\"Training samples: {len(train_df)}\")\nprint(f\"Test samples: {len(test_df)}\")\nprint(f\"Sampling frequencies: {sorted(train_df['fs'].unique())} Hz\")\nprint(f\"ECG leads: {expected_leads}\")\nprint(f\"Image types per training sample: 9\")\nprint(f\"Image types per test sample: 1\")\nprint(f\"Signal duration (training): 10 seconds\")\nprint(f\"Signal duration (test - Lead II): 10 seconds\")\nprint(f\"Signal duration (test - other leads): 2.5 seconds\")\nprint(f\"Total submission rows: {len(sample_submission)}\")\nprint(\"=\" * 80)","metadata":{"_uuid":"0ba9e95f-c97f-4796-8d82-58ae9b812af4","_cell_guid":"362a6f66-bd6f-4027-8011-e0fbffec4c89","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-10-21T18:49:49.442381Z","iopub.execute_input":"2025-10-21T18:49:49.442700Z","iopub.status.idle":"2025-10-21T18:49:49.450196Z","shell.execute_reply.started":"2025-10-21T18:49:49.442673Z","shell.execute_reply":"2025-10-21T18:49:49.449238Z"}},"outputs":[],"execution_count":null}]}