{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":97984,"databundleVersionId":14096757}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install webdataset -q","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#config\npath=r'/kaggle/input/physionet-ecg-image-digitization/train'\noutput_path=r'/kaggle/working/'\noutput_pattern=r'/kaggle/working/folder-%06d.tar'\noutput_pattern_csv=r'/kaggle/working/csv-%06d.tar'\nshards=5000\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**file interpartion**","metadata":{}},{"cell_type":"code","source":"from pathlib import Path  \nimport os\nimport time\nfrom tqdm import tqdm\nPpath=Path(path)\nall_files=[i for i in Ppath.rglob('*') if i.is_file()]\nprint(len(os.listdir(Ppath)))\np_bar=tqdm(all_files,unit=\"flies\",leave=True)\ntotal_size=0\nstart_time=time.time()\nfor i in p_bar:\n    f_size=i.stat().st_size\n    total_size+=f_size\n    p_bar.set_postfix({\n            \"time\": f\"{f_size:.2f}\",\n            \"total_GB\": f\"{total_size/1024**3:.2f}\"\n        })\n    p_bar.update(1)\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CONVERTING INTO .TAR  AND READING THE DATA ","metadata":{}},{"cell_type":"code","source":"import webdataset as wb\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\nfrom torch.utils.data import DataLoader\nimport torchvision.transforms as T","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(path: Path):\n    try:\n        ext=path.suffix.lower()\n        if ext=='.png':\n            img_bytes= path.read_bytes()\n            label = path.parent.name      # folder name = label\n            key = path.stem \n            return {\n            \"__key__\": key,\n            \"png\": img_bytes,\n            \"label\": label,           # <-- FIX: use 'label'\n            \"size\": len(img_bytes)\n        }\n        elif ext=='.csv':\n            csv_text=path.read_bytes()\n            label = path.parent.name      # folder name = label\n            key = path.stem \n                      \n\n            return {\n                \"__key__\": key,\n                \"csv\": csv_text,\n                \"label\": label,           # <-- FIX: use 'label'\n                \"size\": len(csv_text)\n            }\n\n    except Exception as e:\n        print(\"ERROR:\", path, e)\n        return None\n\ndef Tar_conversion(path:Path,shards:int,output_path:str,out_pattern:str,file_pattern:str='*.png',limit:int=10,num_workers=4):\n    all_files=list(i for i in path.rglob(file_pattern) if i.is_file())\n    out_path=Path(output_path)\n    out_path.mkdir(parents=True, exist_ok=True)\n\n    sink= wb.ShardWriter(out_pattern, maxcount=shards)\n    stop=False\n    ts=0\n    with ThreadPoolExecutor(max_workers=num_workers) as executor:\n        futures = {executor.submit(load_data, f): f for f in all_files}\n        for f in as_completed(futures):\n            sample =f.result()\n            # print(sample)\n            if sample is None:\n                continue\n            f_s=sample.pop(\"size\")\n            ts+=f_s/1024**3\n            if ts>limit:\n                print(f'reached lmit{ts}')\n                break\n            sink.write(sample)\n        for f in futures:\n            f.cancel()\n    sink.close()\n    print(\"Tar converison is completed\")\ndef read_data(output_path:str):\n    pass\ndef main():\n    in_path=Path(path)\n    out_path=Path(output_path)\n    Tar_conversion(in_path,shards,out_path,output_pattern)\n    # Tar_conversion(in_path,shards,out_path,output_pattern_csv,file_pattern='*.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"main()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import webdataset as wds\ntransform = T.Compose([\n    T.Resize((720, 720)),\n    T.ToTensor()\n])\ndataset = (\n    wds.WebDataset(\"/kaggle/working/folder-{000000..000003}.tar\",shardshuffle=False)\n    .decode(\"pil\")   # decode PNG → PIL Image\n    .to_tuple(\"png\", \"label\")\n    .map_tuple(transform, lambda x: x)\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfor i, (img, label) in enumerate(dataset):\n    plt.imshow(img.permute(1,2,0).numpy())\n    plt.title(label)\n    plt.axis(\"off\")\n    plt.show()\n\n    if i == 4:   # show first 5 images\n        break\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Main_loader=DataLoader(dataset,batch_size=32,num_workers=2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA\n","metadata":{}},{"cell_type":"code","source":"import random\nimport torch\nfrom tqdm import tqdm\n\ndevice = \"cuda\"\nN = 200\nsamples = []\n\nwith torch.no_grad():\n    pbar = tqdm(Main_loader, desc=\"Sampling Images\", unit=\"batch\")\n\n    for imgs, labels in pbar:\n        # Move entire batch to GPU\n        imgs = imgs.to(device)\n\n        # Random sampling from batch\n        for i in range(len(imgs)):\n            if random.random() < 0.15:\n                samples.append(imgs[i])\n\n            if len(samples) >= N:\n                break\n\n        # Update tqdm with live sample count\n        pbar.set_postfix({\n            \"collected\": len(samples),\n            \"remaining\": max(0, N - len(samples)),\n        })\n\n        if len(samples) >= N:\n            break\n\nprint(f\"Final Samples Collected: {len(samples)}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\nimport numpy as np\nimport cv2 \nfrom concurrent.futures import ThreadPoolExecutor, as_completed","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\nimport numpy as np\n\n# ============================================================\n# 1. Brightness Variation (GPU)\n# ============================================================\ndef brightness_metrics(imgs):\n    gray = 0.299*imgs[:,0] + 0.587*imgs[:,1] + 0.114*imgs[:,2]\n    mean_vals = gray.mean(dim=(1,2))\n    var = mean_vals.var().item()\n    cv = (mean_vals.std()/mean_vals.mean()).item()\n    return var, cv\n\n\n# ============================================================\n# 2. Noise Variation (GPU) using Laplacian filter\n# ============================================================\ndef noise_metrics(imgs):\n    # 3x3 Laplacian kernel\n    kernel = torch.tensor([[0,-1,0],\n                           [-1,4,-1],\n                           [0,-1,0]], \n                           dtype=torch.float32, device=imgs.device).unsqueeze(0).unsqueeze(0)\n\n    gray = 0.299*imgs[:,0:1] + 0.587*imgs[:,1:2] + 0.114*imgs[:,2:3]\n    noise_map = F.conv2d(gray, kernel, padding=1)\n    noise_vals = noise_map.abs().mean(dim=(1,2,3))\n    return noise_vals.var().item(), (noise_vals.std()/noise_vals.mean()).item()\n\n\n# ============================================================\n# 3. Background Variation (GPU)\n# ============================================================\ndef background_metrics(imgs):\n    gray = 0.299*imgs[:,0] + 0.587*imgs[:,1] + 0.114*imgs[:,2]\n    bg_std = gray.std(dim=(1,2))\n    return bg_std.var().item(), (bg_std.std()/bg_std.mean()).item()\n\n\n# ============================================================\n# 4. Curve Thickness Variation (GPU edge detection)\n# ============================================================\ndef thickness_metrics(imgs):\n    gray = 0.299*imgs[:,0:1] + 0.587*imgs[:,1:2] + 0.114*imgs[:,2:3]\n\n    sobel_x = torch.tensor([[-1,0,1], [-2,0,2], [-1,0,1]], \n                            dtype=torch.float32, device=imgs.device).unsqueeze(0).unsqueeze(0)\n    sobel_y = sobel_x.transpose(2,3)\n\n    edges = torch.sqrt(\n        F.conv2d(gray, sobel_x, padding=1)**2 +\n        F.conv2d(gray, sobel_y, padding=1)**2\n    )\n\n    thickness_vals = edges.mean(dim=(1,2,3))\n    return thickness_vals.var().item(), (thickness_vals.std()/thickness_vals.mean()).item()\n\n\n# ============================================================\n# 5. Grid Pattern Strength using FFT (GPU)\n# ============================================================\ndef grid_metrics(imgs):\n    gray = 0.299*imgs[:,0] + 0.587*imgs[:,1] + 0.114*imgs[:,2]\n    fft_vals = torch.fft.fftshift(torch.fft.fft2(gray))  \n    grid_strength = fft_vals.abs().mean(dim=(1,2))\n    return grid_strength.var().item(), (grid_strength.std()/grid_strength.mean()).item()\n\n\n# ============================================================\n# 6. Rotation Variation (GPU PCA-based angle)\n# ============================================================\ndef rotation_metrics(imgs):\n    gray = 0.299*imgs[:,0] + 0.587*imgs[:,1] + 0.114*imgs[:,2]\n    N, H, W = gray.shape\n\n    flat = gray.reshape(N, -1)\n    xs = torch.linspace(0, 1, W, device=imgs.device)\n    ys = torch.linspace(0, 1, H, device=imgs.device)\n    grid_y, grid_x = torch.meshgrid(ys, xs, indexing=\"ij\")\n\n    coords = torch.stack([grid_x.flatten(), grid_y.flatten()], dim=0).unsqueeze(0) \n    coords = coords.repeat(N,1,1)\n\n    weights = flat.unsqueeze(1)\n    cov = torch.matmul(weights * coords, coords.transpose(1,2)) / (H*W)\n\n    eigvals, eigvecs = torch.linalg.eigh(cov)\n    angles = torch.atan2(eigvecs[:,1,1], eigvecs[:,0,1]).abs()\n    return angles.var().item(), (angles.std()/angles.mean()).item()\n\n\n# ============================================================\n# FINAL ANALYZER\n# ============================================================\ndef analyze_dataset_gpu(samples):\n    imgs = torch.stack(samples)  # (N,3,H,W)\n\n    results = {}\n\n    # Brightness\n    b_var, b_cv = brightness_metrics(imgs)\n    results[\"brightness_var\"] = b_var\n    results[\"brightness_cv\"] = b_cv\n\n    # Noise\n    n_var, n_cv = noise_metrics(imgs)\n    results[\"noise_var\"] = n_var\n    results[\"noise_cv\"] = n_cv\n\n    # Background\n    bg_var, bg_cv = background_metrics(imgs)\n    results[\"background_var\"] = bg_var\n    results[\"background_cv\"] = bg_cv\n\n    # Thickness\n    t_var, t_cv = thickness_metrics(imgs)\n    results[\"thickness_var\"] = t_var\n    results[\"thickness_cv\"] = t_cv\n\n    # Grid\n    g_var, g_cv = grid_metrics(imgs)\n    results[\"grid_var\"] = g_var\n    results[\"grid_cv\"] = g_cv\n\n    # Rotation\n    r_var, r_cv = rotation_metrics(imgs)\n    results[\"rotation_var\"] = r_var\n    results[\"rotation_cv\"] = r_cv\n\n    # ---------------------------------------------------------\n    # RECOMMENDED AUGMENTATION RULES BASED ON METRICS\n    # ---------------------------------------------------------\n    aug = []\n\n    # Brightness augmentation\n    if b_cv > 0.10:\n        aug.append(\"RandomBrightnessContrast\")\n    if b_cv > 0.18:\n        aug.append(\"GammaCorrection\")\n\n    # Noise augmentation\n    if n_cv > 0.15:\n        aug.append(\"GaussNoise\")\n    if n_cv > 0.25:\n        aug.append(\"ISONoise\")\n\n    # Background variation\n    if bg_cv > 0.15:\n        aug.append(\"CLAHE\")\n\n    # Thickness variation\n    if t_cv > 0.12:\n        aug.append(\"ShiftScale\")\n\n    # Grid variation\n    if g_cv > 0.20:\n        aug.append(\"ElasticTransform\")\n    if g_cv > 0.30:\n        aug.append(\"GridDistortion\")\n\n    # Rotation\n    if r_cv > 0.15:\n        aug.append(\"ShiftScaleRotate\")\n\n    results[\"recommended_augmentations\"] = aug\n\n    return results\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyze_dataset_gpu(samples)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom PIL import Image, ImageEnhance\nimport matplotlib.pyplot as plt\n\ndef desaturate_red_channel(image):\n    \"\"\"\n    Step 1: Desaturate and lighten red graticule using HSV color space\n    \"\"\"\n    # Convert to HSV\n    hsv = cv2.cvtColor(image, cv2.COLOR_BGR2HSV)\n    \n    # Define red color range (red wraps around in HSV: 0-10 and 170-180)\n    lower_red1 = np.array([0, 50, 50])\n    upper_red1 = np.array([10, 255, 255])\n    lower_red2 = np.array([170, 50, 50])\n    upper_red2 = np.array([180, 255, 255])\n    \n    # Create masks for red regions\n    mask1 = cv2.inRange(hsv, lower_red1, upper_red1)\n    mask2 = cv2.inRange(hsv, lower_red2, upper_red2)\n    red_mask = cv2.bitwise_or(mask1, mask2)\n    \n    # Desaturate red areas (set saturation to minimum)\n    hsv[:, :, 1] = np.where(red_mask > 0, 0, hsv[:, :, 1])\n    \n    # Maximize lightness/value for red areas\n    hsv[:, :, 2] = np.where(red_mask > 0, 255, hsv[:, :, 2])\n    \n    # Convert back to BGR\n    result = cv2.cvtColor(hsv, cv2.COLOR_HSV2BGR)\n    \n    return result\n\ndef apply_smart_blur(image, spatial_radius=10, color_radius=50):\n    \"\"\"\n    Step 2: Apply smart blur to smooth fine grains while preserving edges\n    Similar to Photoshop's Smart Blur filter\n    \"\"\"\n    # Convert to grayscale if needed\n    if len(image.shape) == 3:\n        gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n    else:\n        gray = image.copy()\n    \n    # Apply bilateral filter (preserves edges while smoothing)\n    # This is similar to Smart Blur\n    blurred = cv2.bilateralFilter(gray, d=9, sigmaColor=color_radius, sigmaSpace=spatial_radius)\n    \n    return blurred\n\ndef create_gradient_overlay(image_shape, gradient_type='horizontal'):\n    \"\"\"\n    Step 3: Create gradient layer to balance brightness\n    From black (left) to 50% gray (right)\n    \"\"\"\n    height, width = image_shape[:2]\n    \n    if gradient_type == 'horizontal':\n        # Create horizontal gradient: 0 (black) on left to 128 (50% gray) on right\n        gradient = np.tile(np.linspace(0, 128, width, dtype=np.uint8), (height, 1))\n    else:\n        # Vertical gradient if needed\n        gradient = np.tile(np.linspace(0, 128, height, dtype=np.uint8).reshape(-1, 1), (1, width))\n    \n    return gradient\n\ndef apply_gradient_addition(image, gradient):\n    \"\"\"\n    Add gradient to image to balance brightness (blending mode: Add)\n    \"\"\"\n    # Ensure same dimensions\n    if len(image.shape) == 3:\n        gradient_3ch = cv2.cvtColor(gradient, cv2.COLOR_GRAY2BGR)\n        result = cv2.add(image, gradient_3ch)\n    else:\n        result = cv2.add(image, gradient)\n    \n    return result\n\ndef adjust_levels(image, black_point=50, white_point=200, gamma=1.0):\n    \"\"\"\n    Step 4: Adjust levels to push dark to black and light to white\n    \"\"\"\n    # Normalize to 0-1 range\n    normalized = image.astype(np.float32) / 255.0\n    \n    # Apply levels adjustment\n    # Clip values to black and white points\n    normalized = np.clip((normalized - black_point/255.0) / ((white_point - black_point)/255.0), 0, 1)\n    \n    # Apply gamma correction if needed\n    if gamma != 1.0:\n        normalized = np.power(normalized, gamma)\n    \n    # Convert back to uint8\n    result = (normalized * 255).astype(np.uint8)\n    \n    return result\n\ndef process_image(input_path, output_path=None, show_steps=True):\n    \"\"\"\n    Complete pipeline to remove red graticule from curve image\n    \"\"\"\n    # Read image\n    image = cv2.imread(input_path)\n    if image is None:\n        raise ValueError(f\"Could not read image from {input_path}\")\n    \n    original = image.copy()\n    \n    print(\"Step 1: Desaturating and lightening red graticule...\")\n    step1 = desaturate_red_channel(image)\n    \n    print(\"Step 2: Applying smart blur...\")\n    step2 = apply_smart_blur(step1)\n    \n    print(\"Step 3: Creating and applying gradient overlay...\")\n    gradient = create_gradient_overlay(step2.shape)\n    step3 = apply_gradient_addition(step2, gradient)\n    \n    print(\"Step 4: Adjusting levels...\")\n    step4 = adjust_levels(step3, black_point=30, white_point=220)\n    \n    # Optional: Apply threshold for final cleanup\n    print(\"Step 5: Final thresholding (optional)...\")\n    _, final = cv2.threshold(step4, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    \n    # Save result\n    if output_path:\n        cv2.imwrite(output_path, final)\n        print(f\"Saved result to {output_path}\")\n    \n    # Display steps\n    if show_steps:\n        plt.figure(figsize=(15, 10))\n        \n        plt.subplot(2, 3, 1)\n        plt.imshow(cv2.cvtColor(original, cv2.COLOR_BGR2RGB))\n        plt.title('Original')\n        plt.axis('off')\n        \n        plt.subplot(2, 3, 2)\n        plt.imshow(cv2.cvtColor(step1, cv2.COLOR_BGR2RGB))\n        plt.title('Step 1: Desaturated Red')\n        plt.axis('off')\n        \n        plt.subplot(2, 3, 3)\n        plt.imshow(step2, cmap='gray')\n        plt.title('Step 2: Smart Blur')\n        plt.axis('off')\n        \n        plt.subplot(2, 3, 4)\n        plt.imshow(gradient, cmap='gray')\n        plt.title('Step 3a: Gradient')\n        plt.axis('off')\n        \n        plt.subplot(2, 3, 5)\n        plt.imshow(step3, cmap='gray')\n        plt.title('Step 3b: After Gradient')\n        plt.axis('off')\n        \n        plt.subplot(2, 3, 6)\n        plt.imshow(final, cmap='gray')\n        plt.title('Final Result')\n        plt.axis('off')\n        \n        plt.tight_layout()\n        plt.show()\n    \n    return final\n\n# Example usage\nif __name__ == \"__main__\":\n    # Process your image\n    input_image = \"/kaggle/input/physionet-ecg-image-digitization/train/1006867983/1006867983-000.png\"  # Replace with your image path\n    output_image = \"output_clean_curve.jpg\"\n    \n    try:\n        result = process_image(input_image, output_image, show_steps=True)\n        print(\"Processing complete!\")\n    except Exception as e:\n        print(f\"Error: {e}\")\n        print(\"\\nTo use this script:\")\n        print(\"1. Install required packages: pip install opencv-python numpy pillow matplotlib\")\n        print(\"2. Replace 'input_curve_with_graticule.jpg' with your image path\")\n        print(\"3. Run the script\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img=cv2.imread(r'/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285-0001.png')\nimg=cv2.cvtColor(img,cv2.COLOR_BGR2HSV)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(img)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom skimage.filters import frangi\nimg=cv2.imread(r'/kaggle/input/physionet-ecg-image-digitization/train/735384893/735384893-0003.png')\nimg=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T14:12:43.916963Z","iopub.execute_input":"2025-12-25T14:12:43.917454Z","iopub.status.idle":"2025-12-25T14:12:44.739508Z","shell.execute_reply.started":"2025-12-25T14:12:43.917426Z","shell.execute_reply":"2025-12-25T14:12:44.738654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img2=cv2.imread(r'/kaggle/input/physionet-ecg-image-digitization/train/735384893/735384893-0005.png')\nimg2=cv2.cvtColor(img2,cv2.COLOR_BGR2RGB)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:37:55.172258Z","iopub.execute_input":"2025-12-25T12:37:55.172829Z","iopub.status.idle":"2025-12-25T12:37:55.65844Z","shell.execute_reply.started":"2025-12-25T12:37:55.172805Z","shell.execute_reply":"2025-12-25T12:37:55.657673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gray1=cv2.cvtColor(img,cv2.COLOR_RGB2GRAY)\ngray2=cv2.cvtColor(img2,cv2.COLOR_RGB2GRAY)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:37:56.707636Z","iopub.execute_input":"2025-12-25T12:37:56.707936Z","iopub.status.idle":"2025-12-25T12:37:56.722246Z","shell.execute_reply.started":"2025-12-25T12:37:56.707915Z","shell.execute_reply":"2025-12-25T12:37:56.721509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(np.std(gray1),np.std(gray2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:37:57.73825Z","iopub.execute_input":"2025-12-25T12:37:57.738723Z","iopub.status.idle":"2025-12-25T12:37:57.818244Z","shell.execute_reply.started":"2025-12-25T12:37:57.738701Z","shell.execute_reply":"2025-12-25T12:37:57.817569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 2, figsize=(10, 8))\nax[0,0].imshow(img)\nax[1,0].imshow(img2)\nax[0,1].hist(gray1.ravel(),bins=256,range=(0,256))\nax[1,1].hist(gray2.ravel(),bins=256,range=(0,256))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:37:58.826062Z","iopub.execute_input":"2025-12-25T12:37:58.826321Z","iopub.status.idle":"2025-12-25T12:38:01.507974Z","shell.execute_reply.started":"2025-12-25T12:37:58.826302Z","shell.execute_reply":"2025-12-25T12:38:01.507129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew,kurtosis\nprint(kurtosis(img.ravel(),axis=0,bias=True))\nprint(kurtosis(img2.ravel(),axis=0,bias=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:38:01.508988Z","iopub.execute_input":"2025-12-25T12:38:01.509205Z","iopub.status.idle":"2025-12-25T12:38:02.740336Z","shell.execute_reply.started":"2025-12-25T12:38:01.509189Z","shell.execute_reply":"2025-12-25T12:38:02.739541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gray=cv2.cvtColor(img,cv2.COLOR_BGR2GRAY)\nplt.imshow(gray,cmap='gray')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:38:02.741089Z","iopub.execute_input":"2025-12-25T12:38:02.741445Z","iopub.status.idle":"2025-12-25T12:38:03.216982Z","shell.execute_reply.started":"2025-12-25T12:38:02.741416Z","shell.execute_reply":"2025-12-25T12:38:03.216325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ret, thresh = cv2.threshold(gray,0,255, cv2.THRESH_BINARY+cv2.THRESH_OTSU)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:38:03.218211Z","iopub.execute_input":"2025-12-25T12:38:03.218405Z","iopub.status.idle":"2025-12-25T12:38:03.230696Z","shell.execute_reply.started":"2025-12-25T12:38:03.218389Z","shell.execute_reply":"2025-12-25T12:38:03.229988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(thresh,cmap='gray')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:38:03.231625Z","iopub.execute_input":"2025-12-25T12:38:03.231929Z","iopub.status.idle":"2025-12-25T12:38:03.695Z","shell.execute_reply.started":"2025-12-25T12:38:03.231913Z","shell.execute_reply":"2025-12-25T12:38:03.694287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"blur = cv2.GaussianBlur(gray2, (15,15), 0)\nnorm = cv2.divide(gray2, blur, scale=255)\nkernel = cv2.getStructuringElement(cv2.MORPH_RECT, (7,7))\nridge_u8 = (ridge * 255).astype(np.uint8)\n\n# optional threshold\n_, ridge_bin = cv2.threshold(ridge_u8, 0, 255,\n                              cv2.THRESH_BINARY + cv2.THRESH_OTSU)\nedges = cv2.dilate(ridge_bin, kernel, iterations=1)\n\ncontours, _ = cv2.findContours(edges, cv2.RETR_EXTERNAL,\n                               cv2.CHAIN_APPROX_SIMPLE)\n\ncnt = max(contours, key=cv2.contourArea)\nperi = cv2.arcLength(cnt, True)\napprox = cv2.approxPolyDP(cnt, 0.02 * peri, True)\n\nif len(approx) == 4:\n    pts = approx.reshape(4,2)\n\n    def order_points(pts):\n        s = pts.sum(axis=1)\n        diff = np.diff(pts, axis=1)\n        return np.array([\n            pts[np.argmin(s)],\n            pts[np.argmin(diff)],\n            pts[np.argmax(s)],\n            pts[np.argmax(diff)]\n        ])\n\n    rect = order_points(pts)\n    (tl,tr,br,bl) = rect\n\n    w = int(max(np.linalg.norm(br-bl), np.linalg.norm(tr-tl)))\n    h = int(max(np.linalg.norm(tr-br), np.linalg.norm(tl-bl)))\n\n    dst = np.array([[0,0],[w-1,0],[w-1,h-1],[0,h-1]], dtype=np.float32)\n    M = cv2.getPerspectiveTransform(rect.astype(np.float32), dst)\n    sheet = cv2.warpPerspective(img2, M, (w,h))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:44:06.990671Z","iopub.execute_input":"2025-12-25T12:44:06.991521Z","iopub.status.idle":"2025-12-25T12:44:07.144835Z","shell.execute_reply.started":"2025-12-25T12:44:06.991487Z","shell.execute_reply":"2025-12-25T12:44:07.144197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6,4))\nplt.imshow(cv2.cvtColor(sheet, cv2.COLOR_BGR2RGB))\nplt.axis(\"off\")\nplt.title(\"Extracted ECG Graph Sheet\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:44:15.460054Z","iopub.execute_input":"2025-12-25T12:44:15.460306Z","iopub.status.idle":"2025-12-25T12:44:16.459422Z","shell.execute_reply.started":"2025-12-25T12:44:15.460289Z","shell.execute_reply":"2025-12-25T12:44:16.458636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(sheet)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:44:39.662495Z","iopub.execute_input":"2025-12-25T12:44:39.662824Z","iopub.status.idle":"2025-12-25T12:44:41.098507Z","shell.execute_reply.started":"2025-12-25T12:44:39.6628Z","shell.execute_reply":"2025-12-25T12:44:41.097757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hsv_img=cv2.cvtColor(sheet,cv2.COLOR_RGB2HSV)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:45:55.164694Z","iopub.execute_input":"2025-12-25T12:45:55.165607Z","iopub.status.idle":"2025-12-25T12:45:55.181354Z","shell.execute_reply.started":"2025-12-25T12:45:55.165577Z","shell.execute_reply":"2025-12-25T12:45:55.180737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"low_red1=np.array([0,100,100])\nhigh_red1=np.array([10,255,255])\nlow_red2=np.array([170,100,100])\nhigh_red2=np.array([180,255,255])\nlr=cv2.inRange(hsv_img,low_red1,high_red1)\nhr=cv2.inRange(hsv_img,low_red2,high_red2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:45:57.913927Z","iopub.execute_input":"2025-12-25T12:45:57.914594Z","iopub.status.idle":"2025-12-25T12:45:57.939646Z","shell.execute_reply.started":"2025-12-25T12:45:57.914572Z","shell.execute_reply":"2025-12-25T12:45:57.938892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask=cv2.bitwise_or(lr,hr)\nhsv_img[:,:,1]=np.where(mask>0,0,hsv_img[:,:,1])\nhsv_img[:,:,2]=np.where(mask>0,255,hsv_img[:,:,2])\nfinal_image=cv2.cvtColor(hsv_img,cv2.COLOR_HSV2RGB)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:46:00.654357Z","iopub.execute_input":"2025-12-25T12:46:00.655055Z","iopub.status.idle":"2025-12-25T12:46:00.704889Z","shell.execute_reply.started":"2025-12-25T12:46:00.655033Z","shell.execute_reply":"2025-12-25T12:46:00.704013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(final_image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:46:03.943151Z","iopub.execute_input":"2025-12-25T12:46:03.943827Z","iopub.status.idle":"2025-12-25T12:46:05.397796Z","shell.execute_reply.started":"2025-12-25T12:46:03.943801Z","shell.execute_reply":"2025-12-25T12:46:05.397009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gray=cv2.cvtColor(final_image,cv2.COLOR_RGB2GRAY)\nblur_image= cv2.bilateralFilter(gray, d=5, sigmaColor=50, sigmaSpace=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:46:10.683548Z","iopub.execute_input":"2025-12-25T12:46:10.684106Z","iopub.status.idle":"2025-12-25T12:46:10.731698Z","shell.execute_reply.started":"2025-12-25T12:46:10.684083Z","shell.execute_reply":"2025-12-25T12:46:10.731146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(blur_image,cmap=\"gray\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:46:13.010151Z","iopub.execute_input":"2025-12-25T12:46:13.010761Z","iopub.status.idle":"2025-12-25T12:46:13.858148Z","shell.execute_reply.started":"2025-12-25T12:46:13.010739Z","shell.execute_reply":"2025-12-25T12:46:13.857238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"row,col=blur_image.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:46:17.21705Z","iopub.execute_input":"2025-12-25T12:46:17.217557Z","iopub.status.idle":"2025-12-25T12:46:17.221469Z","shell.execute_reply.started":"2025-12-25T12:46:17.217528Z","shell.execute_reply":"2025-12-25T12:46:17.220713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gradient=np.tile(np.linspace(11,130,col),(row,1)).astype(np.uint8)\nfinal_img=cv2.add(gradient,blur_image)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:48:08.869601Z","iopub.execute_input":"2025-12-25T12:48:08.870191Z","iopub.status.idle":"2025-12-25T12:48:08.900702Z","shell.execute_reply.started":"2025-12-25T12:48:08.870167Z","shell.execute_reply":"2025-12-25T12:48:08.899946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(final_img,cmap='gray')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:48:11.84614Z","iopub.execute_input":"2025-12-25T12:48:11.846394Z","iopub.status.idle":"2025-12-25T12:48:12.674134Z","shell.execute_reply.started":"2025-12-25T12:48:11.846375Z","shell.execute_reply":"2025-12-25T12:48:12.673439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"norm_img=final_img.astype(np.float32)/255","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:47:28.94459Z","iopub.execute_input":"2025-12-25T12:47:28.94521Z","iopub.status.idle":"2025-12-25T12:47:28.967354Z","shell.execute_reply.started":"2025-12-25T12:47:28.945188Z","shell.execute_reply":"2025-12-25T12:47:28.966748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gamma_2=gamma[200:,:]\ncv2.imwrite(\"cv_output.png\",gamma_2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:47:30.533708Z","iopub.execute_input":"2025-12-25T12:47:30.534278Z","iopub.status.idle":"2025-12-25T12:47:30.543121Z","shell.execute_reply.started":"2025-12-25T12:47:30.534258Z","shell.execute_reply":"2025-12-25T12:47:30.542222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(gamma_2,cmap='gray')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:47:35.274781Z","iopub.execute_input":"2025-12-25T12:47:35.275555Z","iopub.status.idle":"2025-12-25T12:47:35.283622Z","shell.execute_reply.started":"2025-12-25T12:47:35.275529Z","shell.execute_reply":"2025-12-25T12:47:35.282778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gamma=np.power(norm_img,3)\nplt.imshow(gamma,cmap='gray')\ncv2.imwrite(\"cv_output.png\",gamma)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T12:47:39.261549Z","iopub.execute_input":"2025-12-25T12:47:39.262187Z","iopub.status.idle":"2025-12-25T12:47:40.232884Z","shell.execute_reply.started":"2025-12-25T12:47:39.26216Z","shell.execute_reply":"2025-12-25T12:47:40.232252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"crop_image=gamma[200:,:]\nplt.imshow(crop_image,cmap='gray')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:21:29.754476Z","iopub.execute_input":"2025-12-13T09:21:29.755266Z","iopub.status.idle":"2025-12-13T09:21:30.187867Z","shell.execute_reply.started":"2025-12-13T09:21:29.755232Z","shell.execute_reply":"2025-12-13T09:21:30.187183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nsm_data=pd.read_csv(r'/kaggle/input/physionet-ecg-image-digitization/train/1006427285/1006427285.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sm_data.dropna(axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data=np.linspace(1,20,128)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# binary_mask: 1 = curve pixels\nrow_density = gamma_2.sum(axis=1)\n\n# smooth\nfrom scipy.ndimage import gaussian_filter1d\nrow_density_smooth = gaussian_filter1d(row_density, sigma=10)\n\n# find valleys → separators\nfrom scipy.signal import find_peaks\nvalleys, _ = find_peaks(-row_density_smooth, distance=50)\n\nvalleys = np.sort(valleys)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:23:36.739914Z","iopub.execute_input":"2025-12-13T09:23:36.740495Z","iopub.status.idle":"2025-12-13T09:23:36.746739Z","shell.execute_reply.started":"2025-12-13T09:23:36.740473Z","shell.execute_reply":"2025-12-13T09:23:36.746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bands = []\nstart = 0\nfor v in valleys:\n    bands.append((start, v))\n    start = v\nbands.append((start, gamma_2.shape[0]))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:23:49.251333Z","iopub.execute_input":"2025-12-13T09:23:49.251858Z","iopub.status.idle":"2025-12-13T09:23:49.25584Z","shell.execute_reply.started":"2025-12-13T09:23:49.251836Z","shell.execute_reply":"2025-12-13T09:23:49.254969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"paths = []\n\nfor (y0, y1) in bands:\n    sub_mask = gamma_2[y0:y1, :]\n    cost = 1.0 - sub_mask.astype(np.float32) + 0.01\n\n    H, W = cost.shape\n    dp = np.full((H, W), np.inf)\n    parent = np.zeros((H, W), dtype=int)\n\n    dp[:, 0] = cost[:, 0]\n\n    for x in range(1, W):\n        for y in range(H):\n            for dy in (-1, 0, 1):\n                py = y + dy\n                if 0 <= py < H:\n                    val = dp[py, x-1] + cost[y, x]\n                    if val < dp[y, x]:\n                        dp[y, x] = val\n                        parent[y, x] = py\n\n    # backtrack\n    y = np.argmin(dp[:, -1])\n    path = []\n    for x in reversed(range(W)):\n        path.append((x, y + y0))\n        y = parent[y, x]\n\n    paths.append(path[::-1])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:23:51.494911Z","iopub.execute_input":"2025-12-13T09:23:51.495612Z","iopub.status.idle":"2025-12-13T09:24:00.906699Z","shell.execute_reply.started":"2025-12-13T09:23:51.495588Z","shell.execute_reply":"2025-12-13T09:24:00.906103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"band_center = (y0 + y1) / 2\ncost += 0.002 * np.abs(np.arange(H)[:, None] - (band_center - y0))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:24:00.90771Z","iopub.execute_input":"2025-12-13T09:24:00.907918Z","iopub.status.idle":"2025-12-13T09:24:00.912264Z","shell.execute_reply.started":"2025-12-13T09:24:00.907902Z","shell.execute_reply":"2025-12-13T09:24:00.911591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"signals = []\n\nfor path in paths:\n    y_coords = np.array([p[1] for p in path])\n    signal = gamma_2.shape[0] - y_coords\n    signals.append(signal)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:24:00.912861Z","iopub.execute_input":"2025-12-13T09:24:00.913015Z","iopub.status.idle":"2025-12-13T09:24:00.928933Z","shell.execute_reply.started":"2025-12-13T09:24:00.913003Z","shell.execute_reply":"2025-12-13T09:24:00.928327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.signal import resample\n\nsignals_resampled = [\n    resample(sig, 5000) for sig in signals\n]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:24:11.807145Z","iopub.execute_input":"2025-12-13T09:24:11.80771Z","iopub.status.idle":"2025-12-13T09:24:11.813735Z","shell.execute_reply.started":"2025-12-13T09:24:11.807688Z","shell.execute_reply":"2025-12-13T09:24:11.813014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pixels_per_mm = 20   # example → YOU must measure this\nmm_per_mV = 10\n\nmV_per_pixel = 1 / (pixels_per_mm * mm_per_mV)\n\nsignals_mV = [sig * mV_per_pixel for sig in signals]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:30:27.662505Z","iopub.execute_input":"2025-12-13T09:30:27.662766Z","iopub.status.idle":"2025-12-13T09:30:27.667419Z","shell.execute_reply.started":"2025-12-13T09:30:27.662749Z","shell.execute_reply":"2025-12-13T09:30:27.666685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fs = 500  # example Hz\nt = np.arange(len(signals_mV[0])) / fs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T09:30:50.325909Z","iopub.execute_input":"2025-12-13T09:30:50.326292Z","iopub.status.idle":"2025-12-13T09:30:50.33023Z","shell.execute_reply.started":"2025-12-13T09:30:50.32627Z","shell.execute_reply":"2025-12-13T09:30:50.329628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Sampling parameters\nfs = 500          # sampling rate (Hz)\nduration = 5      # seconds\nt = np.linspace(0, duration, fs * duration)\n\n# Basic ECG-like waveform (P-QRS-T)\necg = (\n    0.1 * np.sin(2 * np.pi * 1.0 * t) +       # baseline\n    1.2 * np.exp(-((t % 1 - 0.3)/0.02)**2) -  # QRS\n    0.2 * np.exp(-((t % 1 - 0.5)/0.05)**2)    # T wave\n)\n\nplt.plot(t, ecg)\nplt.xlabel(\"Time (s)\")\nplt.ylabel(\"Amplitude (mV)\")\nplt.title(\"Synthetic ECG Signal\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T15:20:08.132691Z","iopub.execute_input":"2025-12-15T15:20:08.133347Z","iopub.status.idle":"2025-12-15T15:20:08.344007Z","shell.execute_reply.started":"2025-12-15T15:20:08.133323Z","shell.execute_reply":"2025-12-15T15:20:08.343449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import neurokit2 as nk\nimport matplotlib.pyplot as plt\n\nfs = 500\nduration = 10  # seconds\n\necg = nk.ecg_simulate(\n    duration=duration,\n    sampling_rate=fs,\n    heart_rate=72,\n    noise=0.01\n)\n\nt = np.arange(len(ecg)) / fs\n\nplt.plot(t, ecg)\nplt.xlabel(\"Time (s)\")\nplt.ylabel(\"Amplitude (mV)\")\nplt.title(\"Simulated ECG (NeuroKit)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T15:21:12.410018Z","iopub.execute_input":"2025-12-15T15:21:12.410345Z","iopub.status.idle":"2025-12-15T15:21:14.282412Z","shell.execute_reply.started":"2025-12-15T15:21:12.410315Z","shell.execute_reply":"2025-12-15T15:21:14.281552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install neurokit2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T15:21:05.162431Z","iopub.execute_input":"2025-12-15T15:21:05.162701Z","iopub.status.idle":"2025-12-15T15:21:09.935889Z","shell.execute_reply.started":"2025-12-15T15:21:05.162678Z","shell.execute_reply":"2025-12-15T15:21:09.934874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}