{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Digitizing Waveform From ECG Image wo/Training**","metadata":{}},{"cell_type":"markdown","source":"\n## **def analyze_and_split(img):**\n\nThis function processes an ECG waveform image and extracts numerical data from it through several stages:\n\n### 1. **Image Preprocessing**\n- **Grayscale Conversion**: Converts the color image to grayscale for simpler processing\n- **Gaussian Blur**: Applies smoothing to reduce noise while preserving important edges\n\n### 2. **Grid Line Removal**\n- **HSV Color Space Conversion**: Converts to HSV for better color segmentation\n- **Red Color Masking**: Identifies and creates a mask for red grid lines\n- **Inpainting**: Removes the detected grid lines while preserving the waveform\n\n### 3. **Waveform Segmentation**\n- **Binarization**: Uses Otsu's thresholding to create a binary image (waveform vs background)\n- **Contour Detection**: Finds all external contours in the image\n- **Main Waveform Selection**: Selects the largest contour as the primary ECG waveform\n\n### 4. **Coordinate Processing**\n- **Point Extraction**: Extracts X-Y coordinates from the contour\n- **Y-axis Inversion**: Flips the Y-axis since image coordinates increase downward\n- **Normalization**: \n  - Shifts Y-values to start at zero\n  - Scales Y-values to range [0, 1] for standardization\n\n### 5. **Resampling and Cleaning**\n- **Interpolation**: Resamples the waveform to exactly 10,000 points for consistency\n- **Endpoint Trimming**: Removes 240 points from start and 140 points from end to eliminate edge artifacts\n- **Final Length**: Results in 9,620 clean, uniformly spaced data points\n\n### 6. **Lead Separation**\n- **Segmentation**: Splits the continuous waveform into 4 equal segments representing different ECG leads\n- **Segment Length**: Each lead contains 2,500 data points\n- **Zero-padding**: Ensures consistent length across all leads if needed\n\n### 7. **Output and Visualization**\n- **DataFrame Creation**: Organizes the extracted data into a pandas DataFrame with separate columns for each lead\n- **Visualization**: Plots all 4 leads with grid lines and point count annotations\n- **Statistics Display**: Shows the number of data points extracted for each lead\n","metadata":{}},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image \nfrom IPython.display import display\nimport io","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:08.745989Z","iopub.execute_input":"2025-10-28T12:31:08.746632Z","iopub.status.idle":"2025-10-28T12:31:10.311707Z","shell.execute_reply.started":"2025-10-28T12:31:08.746605Z","shell.execute_reply":"2025-10-28T12:31:10.310728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Load Image ===\nimg_path = \"/kaggle/input/physionet-ecg-image-digitization/test/1053922973.png\"\nimg = cv2.imread(img_path)\nimg1 = img[530:840]\nimg2 = img[840:1170]\nimg3 = img[1130:1420]\nimg4 = img[1400:1600]\nprint(img.shape)\nplt.imshow(img)\nplt.show()\nplt.imshow(img1)\nplt.show()\nplt.imshow(img2)\nplt.show()\nplt.imshow(img3)\nplt.show()\nplt.imshow(img4)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:08.745989Z","iopub.execute_input":"2025-10-28T12:31:08.746632Z","iopub.status.idle":"2025-10-28T12:31:10.311707Z","shell.execute_reply.started":"2025-10-28T12:31:08.746605Z","shell.execute_reply":"2025-10-28T12:31:10.310728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test=pd.read_csv('/kaggle/input/physionet-ecg-image-digitization/test.csv')\ndisplay(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:10.313603Z","iopub.execute_input":"2025-10-28T12:31:10.313992Z","iopub.status.idle":"2025-10-28T12:31:10.327948Z","shell.execute_reply.started":"2025-10-28T12:31:10.313962Z","shell.execute_reply":"2025-10-28T12:31:10.326868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission=pd.read_parquet('/kaggle/input/physionet-ecg-image-digitization/sample_submission.parquet')\ndisplay(submission)\nsubmission['idi']=submission['id'].apply(lambda x:x.split('_')[0])\nsubmission['index']=submission['id'].apply(lambda x:x.split('_')[1])\nsubmission['lead']=submission['id'].apply(lambda x:x.split('_')[2])\ndisplay(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:10.329065Z","iopub.execute_input":"2025-10-28T12:31:10.329812Z","iopub.status.idle":"2025-10-28T12:31:10.457221Z","shell.execute_reply.started":"2025-10-28T12:31:10.329786Z","shell.execute_reply":"2025-10-28T12:31:10.456329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_and_split(img):\n    # === Preprocessing ===\n    # Convert to grayscale\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    # Apply Gaussian blur\n    blur = cv2.GaussianBlur(gray, (5, 5), 0)\n    \n    # === Grid Removal ===\n    # Convert to HSV color space\n    hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    # Define range for red color (assuming grid lines are red)\n    lower_red = np.array([0, 80, 80])\n    upper_red = np.array([20, 255, 255])\n    # Create mask for red color\n    mask_red = cv2.inRange(hsv, lower_red, upper_red)\n    # Inpaint the grayscale image to remove the grid\n    gray_no_grid = cv2.inpaint(gray, mask_red, 3, cv2.INPAINT_TELEA)\n    \n    # === Binarization (Thresholding) ===\n    # Apply Otsu's thresholding with inverse binary\n    _, binary = cv2.threshold(gray_no_grid, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n    \n    # === Waveform Contour Extraction ===\n    # Find all external contours\n    contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    # Select the contour with the largest area (assumed to be the main waveform)\n    wave = max(contours, key=cv2.contourArea)\n    \n    # === Coordinate Extraction and Preprocessing ===\n    # Squeeze the contour array to remove single-dimensional entries\n    wave_points = np.squeeze(wave)\n    # Sort points by X-coordinate\n    wave_points = wave_points[np.argsort(wave_points[:,0])]\n    \n    # Extract X and Y coordinates\n    x = wave_points[:,0]\n    # Invert Y-axis (since pixel Y increases downwards)\n    y = -wave_points[:,1]\n    # Normalize Y: shift to start at zero\n    y = y - np.min(y)\n    # Normalize Y: scale to a maximum of 1\n    y = y / np.max(y)\n    \n    # === Resampling to 5500 Points ===\n    # Create new X-coordinates for resampling\n    remove_start = 240  # Number of points to remove from the start\n    remove_end = 140    # Number of points to remove from the end \n    x_new = np.linspace(x.min(), x.max(), 10000 +remove_start+remove_end)\n    # Interpolate Y-coordinates\n    y_interp = np.interp(x_new, x, y)\n    y_clean = y_interp[remove_start:- remove_end]\n    # ------------------------------------------------------------------\n    \n    # ------------------------------------------------------------------\n    # === Split into 4 Leads (1100 Points Each) ===\n    segment_length = 2500  \n    leads = []\n    \n    # Total available points for splitting: 4400\n    for i in range(4):\n        start_idx = i * segment_length\n        end_idx = (i + 1) * segment_length\n        \n        if end_idx <= len(y_clean):\n            lead_data = y_clean[start_idx:end_idx]\n            # Zero-padding if necessary (should not occur here)\n            if len(lead_data) < segment_length:\n                lead_data = np.pad(lead_data, (0, segment_length - len(lead_data)), 'constant')\n            leads.append(lead_data)\n    # ------------------------------------------------------------------\n    \n    # === Create DataFrame ===\n    df_leads = pd.DataFrame()\n    for i, lead in enumerate(leads):\n        df_leads[f'lead_{i+1}'] = lead\n    \n    # === Visualization (X-axis is index number) ===\n    plt.figure(figsize=(15, 10))\n    for i in range(4):\n        plt.subplot(2, 2, i+1)\n        # X-axis defaults to index 0, 1, 2, ...\n        plt.plot(df_leads[f'lead_{i+1}'])\n        plt.title(f\"Lead {i+1} - {segment_length} Points\")\n        plt.xlabel(\"Data Point Index\")\n        plt.ylabel(\"Normalized Voltage\")\n        plt.grid(True, alpha=0.3)\n        \n        # Display the number of data points\n        plt.text(0.02, 0.98, f'Points: {len(df_leads[f\"lead_{i+1}\"])}', \n                 transform=plt.gca().transAxes, verticalalignment='top',\n                 bbox=dict(boxstyle='round', facecolor='white', alpha=0.8))\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # === Display Basic Statistics for Each Lead ===\n    print(\"Number of data points for each lead:\")\n    for i in range(4):\n        print(f\"Lead {i+1}: {len(leads[i])} points\")\n    \n    return df_leads\n\n# Example Usage\n# df_ecg = analyze_and_split(img) \n# print(f\"Overall Data Shape: {df_ecg.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:10.458215Z","iopub.execute_input":"2025-10-28T12:31:10.458496Z","iopub.status.idle":"2025-10-28T12:31:10.473642Z","shell.execute_reply.started":"2025-10-28T12:31:10.458471Z","shell.execute_reply":"2025-10-28T12:31:10.472465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_and_split_for2(img):\n    # === Preprocessing ===\n    # Convert to grayscale\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    # Apply Gaussian blur\n    blur = cv2.GaussianBlur(gray, (5, 5), 0)\n    \n    # === Grid Removal ===\n    # Convert to HSV color space\n    hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    # Define range for red color (assuming grid lines are red)\n    lower_red = np.array([0, 80, 80])\n    upper_red = np.array([20, 255, 255])\n    # Create mask for red color\n    mask_red = cv2.inRange(hsv, lower_red, upper_red)\n    # Inpaint the grayscale image to remove the grid\n    gray_no_grid = cv2.inpaint(gray, mask_red, 3, cv2.INPAINT_TELEA)\n    \n    # === Binarization (Thresholding) ===\n    # Apply Otsu's thresholding with inverse binary\n    _, binary = cv2.threshold(gray_no_grid, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n    \n    # === Waveform Contour Extraction ===\n    # Find all external contours\n    contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    # Select the contour with the largest area (assumed to be the main waveform)\n    wave = max(contours, key=cv2.contourArea)\n    \n    # === Coordinate Extraction and Preprocessing ===\n    # Squeeze the contour array to remove single-dimensional entries\n    wave_points = np.squeeze(wave)\n    # Sort points by X-coordinate\n    wave_points = wave_points[np.argsort(wave_points[:,0])]\n    \n    # Extract X and Y coordinates\n    x = wave_points[:,0]\n    # Invert Y-axis (since pixel Y increases downwards)\n    y = -wave_points[:,1]\n    # Normalize Y: shift to start at zero\n    y = y - np.min(y)\n    # Normalize Y: scale to a maximum of 1\n    y = y / np.max(y)\n    \n    # === Resampling to 10000 Points ===\n    # Create new X-coordinates for resampling\n    remove_start = 240  # Number of points to remove from the start\n    remove_end = 140    # Number of points to remove from the end \n    x_new = np.linspace(x.min(), x.max(), 10000+remove_start+remove_end)\n    # Interpolate Y-coordinates\n    y_interp = np.interp(x_new, x, y)\n    y_clean = y_interp[remove_start:-remove_end]\n    # ------------------------------------------------------------------\n    \n    # === Create Single Lead DataFrame ===\n    df_lead = pd.DataFrame({'lead_1': y_clean})\n    \n    # === Visualization ===\n    plt.figure(figsize=(12, 6))\n    plt.plot(df_lead['lead_1'])\n    plt.title(f\"Single Lead - {len(y_clean)} Points\")\n    plt.xlabel(\"Data Point Index\")\n    plt.ylabel(\"Normalized Voltage\")\n    plt.grid(True, alpha=0.3)\n    \n    # Display the number of data points\n    plt.text(0.02, 0.98, f'Points: {len(df_lead[\"lead_1\"])}', \n             transform=plt.gca().transAxes, verticalalignment='top',\n             bbox=dict(boxstyle='round', facecolor='white', alpha=0.8))\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # === Display Basic Statistics ===\n    print(f\"Number of data points: {len(y_clean)}\")\n    print(f\"Data range: {np.min(y_clean):.4f} to {np.max(y_clean):.4f}\")\n    print(f\"Mean: {np.mean(y_clean):.4f}, Std: {np.std(y_clean):.4f}\")\n    \n    return df_lead","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:10.476502Z","iopub.execute_input":"2025-10-28T12:31:10.476807Z","iopub.status.idle":"2025-10-28T12:31:10.497602Z","shell.execute_reply.started":"2025-10-28T12:31:10.476786Z","shell.execute_reply":"2025-10-28T12:31:10.496601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#I,aVR,V1,V4\ndf1=analyze_and_split(img1)\ndisplay(df1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:10.551692Z","iopub.execute_input":"2025-10-28T12:31:10.551986Z","iopub.status.idle":"2025-10-28T12:31:11.566865Z","shell.execute_reply.started":"2025-10-28T12:31:10.551965Z","shell.execute_reply":"2025-10-28T12:31:11.565796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"    mask_i = (submission['idi'] == '1053922973') & (submission['lead'] == 'I')\n    submission.loc[mask_i, 'value'] = df1.loc[:, 'lead_1'].values\n    \n    mask_avr = (submission['idi'] == '1053922973') & (submission['lead'] == 'aVR')\n    submission.loc[mask_avr, 'value'] = df1.loc[:, 'lead_2'].values\n    \n    mask_v1 = (submission['idi'] == '1053922973') & (submission['lead'] == 'V1')\n    submission.loc[mask_v1, 'value'] = df1.loc[:, 'lead_3'].values\n    \n    mask_v4 = (submission['idi'] == '1053922973') & (submission['lead'] == 'V4')\n    submission.loc[mask_v4, 'value'] = df1.loc[:, 'lead_4'].values","metadata":{"execution":{"iopub.status.busy":"2025-10-28T12:31:10.551692Z","iopub.execute_input":"2025-10-28T12:31:10.551986Z","iopub.status.idle":"2025-10-28T12:31:11.566865Z","shell.execute_reply.started":"2025-10-28T12:31:10.551965Z","shell.execute_reply":"2025-10-28T12:31:11.565796Z"}}},{"cell_type":"code","source":"#II,aVL,V2,V5\ndf2=analyze_and_split(img2)\ndisplay(df2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:11.61158Z","iopub.status.idle":"2025-10-28T12:31:11.611883Z","shell.execute_reply.started":"2025-10-28T12:31:11.611745Z","shell.execute_reply":"2025-10-28T12:31:11.611759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"    mask_avl = (submission['idi'] == '1053922973') & (submission['lead'] == 'aVL')\n    submission.loc[mask_avl, 'value'] = df2['lead_2'].values\n    \n    mask_v2 = (submission['idi'] == '1053922973') & (submission['lead'] == 'V2')\n    submission.loc[mask_v2, 'value'] = df2['lead_3'].values\n    \n    mask_v5 = (submission['idi'] == '1053922973') & (submission['lead'] == 'V5')\n    submission.loc[mask_v5, 'value'] = df2['lead_4'].values","metadata":{"execution":{"iopub.status.busy":"2025-10-28T12:31:11.61158Z","iopub.status.idle":"2025-10-28T12:31:11.611883Z","shell.execute_reply.started":"2025-10-28T12:31:11.611745Z","shell.execute_reply":"2025-10-28T12:31:11.611759Z"}}},{"cell_type":"code","source":"#III,aVF,V3,V6\ndf3=analyze_and_split(img3)\ndisplay(df3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:11.61331Z","iopub.status.idle":"2025-10-28T12:31:11.613605Z","shell.execute_reply.started":"2025-10-28T12:31:11.613463Z","shell.execute_reply":"2025-10-28T12:31:11.613475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"    mask_iii = (submission['idi'] == '1053922973') & (submission['lead'] == 'III')\n    submission.loc[mask_iii, 'value'] = df3['lead_1'].values\n    \n    mask_avf = (submission['idi'] == '1053922973') & (submission['lead'] == 'aVF')\n    submission.loc[mask_avf, 'value'] = df3['lead_2'].values\n    \n    mask_v3 = (submission['idi'] == '1053922973') & (submission['lead'] == 'V3')\n    submission.loc[mask_v3, 'value'] = df3['lead_3'].values\n    \n    mask_v6 = (submission['idi'] == '1053922973') & (submission['lead'] == 'V6')\n    submission.loc[mask_v6, 'value'] = df3['lead_4'].values","metadata":{"execution":{"iopub.status.busy":"2025-10-28T12:31:11.61331Z","iopub.status.idle":"2025-10-28T12:31:11.613605Z","shell.execute_reply.started":"2025-10-28T12:31:11.613463Z","shell.execute_reply":"2025-10-28T12:31:11.613475Z"}}},{"cell_type":"code","source":"#II\ndf4=analyze_and_split_for2(img4)\ndisplay(df4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:11.614519Z","iopub.status.idle":"2025-10-28T12:31:11.614884Z","shell.execute_reply.started":"2025-10-28T12:31:11.614703Z","shell.execute_reply":"2025-10-28T12:31:11.614718Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"    mask_ii = (submission['idi'] == '1053922973') & (submission['lead'] == 'II')\n    submission.loc[mask_ii, 'value'] = df4['lead_1'].values","metadata":{"execution":{"iopub.status.busy":"2025-10-28T12:31:11.614519Z","iopub.status.idle":"2025-10-28T12:31:11.614884Z","shell.execute_reply.started":"2025-10-28T12:31:11.614703Z","shell.execute_reply":"2025-10-28T12:31:11.614718Z"}}},{"cell_type":"code","source":"all_mappings = [\n    (df1, {'I': 'lead_1', 'aVR': 'lead_2', 'V1': 'lead_3', 'V4': 'lead_4'}),\n    (df2, {'aVL': 'lead_2', 'V2': 'lead_3', 'V5': 'lead_4'}),\n    (df3, {'III': 'lead_1', 'aVF': 'lead_2', 'V3': 'lead_3', 'V6': 'lead_4'}),\n    (df4, {'II': 'lead_1'})\n]\n\nfor df, mapping in all_mappings:\n    for lead_name, df_column in mapping.items():\n        mask = (submission['idi'] == '1053922973') & (submission['lead'] == lead_name)\n        submission.loc[mask, 'value'] = df[df_column].values","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(submission[submission['idi']=='1053922973'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T12:31:11.615786Z","iopub.status.idle":"2025-10-28T12:31:11.61603Z","shell.execute_reply.started":"2025-10-28T12:31:11.615909Z","shell.execute_reply":"2025-10-28T12:31:11.61592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}