{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":97984,"databundleVersionId":14096757},{"sourceType":"datasetVersion","sourceId":15433411,"datasetId":9872944,"databundleVersionId":16352653},{"sourceType":"datasetVersion","sourceId":15585543,"datasetId":9971684,"databundleVersionId":16517673},{"sourceType":"datasetVersion","sourceId":15590249,"datasetId":9975170,"databundleVersionId":16522655},{"sourceType":"datasetVersion","sourceId":15593434,"datasetId":9977446,"databundleVersionId":16526018},{"sourceType":"datasetVersion","sourceId":15605581,"datasetId":9986458,"databundleVersionId":16539019}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"BATCHES = {\n    'batch1': '/kaggle/input/datasets/tylerde/ecg-training-data-0001-only/data',\n    'batch2': '/kaggle/input/datasets/tylerde/ecg-training-data-0003-0005-0006/data',\n    'batch3': '/kaggle/input/datasets/tylerde/ecg-training-data-0004-0009/data',\n    'batch4': '/kaggle/input/datasets/tylerde/ecg-training-data-0010-0011/data',\n    'batch5': '/kaggle/input/datasets/tylerde/ecg-training-data-0012/data',\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T20:17:09.979570Z","iopub.execute_input":"2026-04-08T20:17:09.979932Z","iopub.status.idle":"2026-04-08T20:17:09.986404Z","shell.execute_reply.started":"2026-04-08T20:17:09.979906Z","shell.execute_reply":"2026-04-08T20:17:09.984413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === ecg0-verify-overlay-all-batches ===\n# Visually verify mask alignment across all datasets\n# Attach all available batch datasets before running\n\nimport os, cv2, numpy as np, pandas as pd\nimport matplotlib.pyplot as plt\n\nZERO_MV = [703.5, 987.5, 1271.5, 1531.5]\nW_SIZE = 240\n\ndef load_sparse_mask(filepath):\n    data = np.load(filepath)\n    shape = tuple(data['shape'])\n    mask = np.zeros(shape, dtype=np.float32)\n    for i in range(shape[0]):\n        mask[i, data[f'ch{i}_y'], data[f'ch{i}_x']] = data[f'ch{i}_v']\n    return mask\n\n\ndef verify_sample(data_dir, sid, type_id, axes_row, title):\n    \"\"\"Overlay mask on one image, draw across 4 axes.\"\"\"\n    img_path = f'{data_dir}/rectified/{sid}-{type_id}.rect.png'\n    mask_path = f'{data_dir}/masks/{sid}.mask-coo.npz'\n\n    if not os.path.exists(img_path):\n        for ax in axes_row:\n            ax.text(0.5, 0.5, 'IMAGE NOT FOUND', ha='center', va='center')\n            ax.set_title(title)\n        return\n\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (5600, 1700), interpolation=cv2.INTER_LINEAR)\n    full_mask = load_sparse_mask(mask_path)\n\n    for row_idx in range(4):\n        zmv = int(ZERO_MV[row_idx])\n        h0, h1 = zmv - W_SIZE, zmv + W_SIZE\n\n        row_img = img[h0:h1, :, :].copy()\n        m = full_mask[row_idx, h0:h1, :]\n        H, W = m.shape\n        y_idx = np.arange(H, dtype=np.float32)[:, None]\n        denom = m.sum(axis=0) + 1e-8\n        y_pos = (m * y_idx).sum(axis=0) / denom\n\n        for x in range(301, 5301):\n            y = int(round(y_pos[x]))\n            if 0 <= y < H:\n                row_img[max(0, y-3):min(H, y+4), x] = [0, 0, 255]\n\n        axes_row[row_idx].imshow(row_img)\n        axes_row[row_idx].set_title(f'Row {row_idx}')\n\n    axes_row[0].set_ylabel(title, fontsize=10, rotation=0, labelpad=80, va='center')\n\n\n# === Load all fold CSVs and pick samples ===\nfor batch_name, data_dir in BATCHES.items():\n    if not os.path.exists(data_dir):\n        print(f'Skipping {batch_name}: not found')\n        continue\n\n    fold_df = pd.read_csv(f'{data_dir}/train_fold.csv', dtype={'id': str, 'type_id': str})\n    type_ids = fold_df['type_id'].unique()\n    print(f'\\n{batch_name}: {len(fold_df)} samples, types: {type_ids}')\n\n    # Pick 3 random samples per type_id\n    np.random.seed(42)\n    for tid in sorted(type_ids):\n        subset = fold_df[fold_df['type_id'] == tid]\n        sample_indices = np.random.choice(len(subset), min(3, len(subset)), replace=False)\n\n        for si in sample_indices:\n            row = subset.iloc[si]\n            sid = str(row['id'])\n            title = f'{batch_name}\\n{tid}\\n{sid}'\n\n            fig, axes = plt.subplots(4, 1, figsize=(20, 12))\n            verify_sample(data_dir, sid, tid, axes, title)\n            plt.suptitle(f'{batch_name} | Segment {tid} | ID {sid}', fontsize=14)\n            plt.tight_layout()\n            plt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-08T20:10:43.781091Z","iopub.execute_input":"2026-04-08T20:10:43.781417Z","iopub.status.idle":"2026-04-08T20:11:35.977307Z","shell.execute_reply.started":"2026-04-08T20:10:43.781383Z","shell.execute_reply":"2026-04-08T20:11:35.976300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Verify completeness ===\nimport os, pandas as pd\n\nfor name, path in BATCHES.items():\n    print(f'\\n=== {name} ===')\n    if not os.path.exists(path):\n        print('  NOT FOUND!')\n        continue\n\n    fold_df = pd.read_csv(f'{path}/train_fold.csv', dtype={'id': str, 'type_id': str})\n    print(f'  Fold CSV rows: {len(fold_df)}')\n    print(f'  Types: {fold_df[\"type_id\"].unique()}')\n\n    # Check rectified images\n    missing_img = 0\n    for _, row in fold_df.iterrows():\n        img_path = f'{path}/rectified/{row[\"id\"]}-{row[\"type_id\"]}.rect.png'\n        if not os.path.exists(img_path):\n            missing_img += 1\n    print(f'  Rectified images: {len(fold_df) - missing_img}/{len(fold_df)} found, {missing_img} missing')\n\n    # Check masks\n    unique_ids = fold_df['id'].unique()\n    missing_mask = sum(1 for sid in unique_ids if not os.path.exists(f'{path}/masks/{sid}.mask-coo.npz'))\n    print(f'  Masks: {len(unique_ids) - missing_mask}/{len(unique_ids)} found, {missing_mask} missing')\n\n    # Check ECG CSVs\n    missing_csv = sum(1 for sid in unique_ids if not os.path.exists(f'{path}/ecg_csv/{sid}.csv'))\n    print(f'  ECG CSVs: {len(unique_ids) - missing_csv}/{len(unique_ids)} found, {missing_csv} missing')\n\n    # Folder sizes\n    for folder in ['rectified', 'masks', 'ecg_csv']:\n        folder_path = f'{path}/{folder}'\n        if os.path.exists(folder_path):\n            n_files = len(os.listdir(folder_path))\n            size = sum(os.path.getsize(os.path.join(folder_path, f)) for f in os.listdir(folder_path))\n            print(f'  {folder}: {n_files} files, {size/1e9:.2f} GB')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T20:17:20.634869Z","iopub.execute_input":"2026-04-08T20:17:20.635419Z","iopub.status.idle":"2026-04-08T20:18:29.230436Z","shell.execute_reply.started":"2026-04-08T20:17:20.635371Z","shell.execute_reply":"2026-04-08T20:18:29.228870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}