{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":22990,"databundleVersionId":2048213,"sourceType":"competition"}],"dockerImageVersionId":31236,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install dependencies\n!pip install -q ultralytics opencv-python-headless rasterio shapely\n\nimport os\nimport cv2\nimport gc\nimport numpy as np\nimport pandas as pd\nimport rasterio\nfrom rasterio.windows import Window\nfrom tqdm.notebook import tqdm\nfrom shapely.geometry import Polygon\nfrom ultralytics import YOLO\nfrom sklearn.model_selection import train_test_split\nimport shutil\n\n# Configuration\nDATA_ROOT = \"/kaggle/input/hubmap-kidney-segmentation\"\nWORK_DIR = \"/kaggle/working/hubmap_yolo\"\n\n# Optimized for speed on Kaggle T4/P100\nPATCH_SIZE = 512\nSTRIDE = 512     # Set to 512 (no overlap) for fastest training. Set to 256 for better accuracy but 4x slower.\nMIN_TISSUE = 0.05 # Drop empty patches\nBATCH_SIZE = 32   # Increased from 8\nEPOCHS = 30      \n\n# Create Directories\nfor split in ['train', 'val']:\n    os.makedirs(f\"{WORK_DIR}/{split}/images\", exist_ok=True)\n    os.makedirs(f\"{WORK_DIR}/{split}/labels\", exist_ok=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle2mask(mask_rle, shape):\n    \"\"\"Decodes RLE string to numpy array.\"\"\"\n    if pd.isna(mask_rle):\n        return np.zeros(shape, dtype=np.uint8)\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape, order='F')\n\ndef tissue_ratio(patch):\n    \"\"\"Calculates the percentage of tissue in a patch.\"\"\"\n    if len(patch.shape) == 2 or patch.shape[-1] == 1:\n        gray = patch\n    else:\n        gray = cv2.cvtColor(patch, cv2.COLOR_RGB2GRAY)\n    \n    # Otsu's thresholding\n    _, mask = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    # Count non-background pixels (assuming background is light/white in H&E usually, \n    # but Otsu finds the separator. For HuBMAP, background is often white).\n    # Inverted check: tissue is usually darker.\n    tissue_pixels = np.sum(mask == 0) \n    return tissue_pixels / mask.size\n\ndef mask_to_polygons(mask):\n    \"\"\"Converts binary mask to normalized polygons for YOLO.\"\"\"\n    contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    polygons = []\n    for cnt in contours:\n        if cv2.contourArea(cnt) < 200: # Filter tiny noise\n            continue\n        cnt = cnt.squeeze()\n        if len(cnt.shape) < 2 or len(cnt) < 6: # Filter invalid lines\n            continue\n        polygons.append(cnt)\n    return polygons","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load metadata\ndf_masks = pd.read_csv(f\"{DATA_ROOT}/train.csv\").set_index('id')\nimage_ids = [f.split(\".\")[0] for f in os.listdir(f\"{DATA_ROOT}/train\") if f.endswith(\".tiff\")]\n\n# Split by IMAGE ID, not patch (Prevents data leakage)\ntrain_ids, val_ids = train_test_split(image_ids, test_size=0.1, random_state=42)\nsplit_map = {img_id: 'train' for img_id in train_ids}\nsplit_map.update({img_id: 'val' for img_id in val_ids})\n\nprint(f\"Processing {len(train_ids)} Train images and {len(val_ids)} Validation images...\")\n\nfor img_id in tqdm(image_ids):\n    if img_id not in df_masks.index: continue\n    \n    split = split_map[img_id]\n    img_path = f\"{DATA_ROOT}/train/{img_id}.tiff\"\n    \n    with rasterio.open(img_path) as src:\n        h, w = src.height, src.width\n        # Load mask only once per image\n        rle = df_masks.loc[img_id, 'encoding']\n        full_mask = rle2mask(rle, (h, w))\n        \n        # Grid loop\n        for y in range(0, h - PATCH_SIZE, STRIDE):\n            for x in range(0, w - PATCH_SIZE, STRIDE):\n                \n                # 1. Read Patch\n                window = Window(x, y, PATCH_SIZE, PATCH_SIZE)\n                patch = src.read(window=window)\n                patch = np.moveaxis(patch, 0, -1) # (C,H,W) -> (H,W,C)\n                \n                # 2. Tissue Check (Skip empty background)\n                if tissue_ratio(patch) < MIN_TISSUE:\n                    continue\n                \n                # 3. Handle Mask Patch\n                mask_patch = full_mask[y:y+PATCH_SIZE, x:x+PATCH_SIZE]\n                \n                # 4. Prepare Filename\n                fname = f\"{img_id}_{x}_{y}\"\n                save_path_img = f\"{WORK_DIR}/{split}/images/{fname}.jpg\"\n                save_path_lbl = f\"{WORK_DIR}/{split}/labels/{fname}.txt\"\n                \n                # 5. Save Image (Force BGR for OpenCV/YOLO)\n                if len(patch.shape) == 2 or patch.shape[-1] == 1:\n                    save_img = cv2.cvtColor(patch, cv2.COLOR_GRAY2BGR)\n                else:\n                    save_img = cv2.cvtColor(patch, cv2.COLOR_RGB2BGR)\n                \n                cv2.imwrite(save_path_img, save_img)\n                \n                # 6. Save Labels\n                polys = mask_to_polygons(mask_patch)\n                with open(save_path_lbl, \"w\") as f:\n                    if polys:\n                        for poly in polys:\n                            # Normalize coordinates (0-1)\n                            poly = poly.astype(float)\n                            poly[:, 0] /= PATCH_SIZE\n                            poly[:, 1] /= PATCH_SIZE\n                            \n                            # Flatten and write\n                            coords = \" \".join(map(str, poly.flatten()))\n                            f.write(f\"0 {coords}\\n\")\n                    import random\n                    # If the patch has NO glomeruli (background), drop 90% of them\n                    if not polys:\n                        if random.random() > 0.10: # Keep only 10% of empty background patches\n                            continue\n                    \n        # Cleanup memory immediately\n        del full_mask\n        gc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_content = f\"\"\"\npath: {WORK_DIR}\ntrain: train/images\nval: val/images\n\nnc: 1\nnames: ['glomerulus']\n\"\"\"\n\nwith open(f\"{WORK_DIR}/data.yaml\", \"w\") as f:\n    f.write(yaml_content)\n\nprint(\"Data YAML created.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load model\nmodel = YOLO(\"yolo11n-seg.pt\") \n\n# Train\nresults = model.train(\n    data=f\"{WORK_DIR}/data.yaml\",\n    epochs=EPOCHS,\n    imgsz=PATCH_SIZE,\n    batch=BATCH_SIZE,      # Higher batch size for speed\n    patience=15,            # Stop early if no improvement\n    optimizer=\"AdamW\",\n    lr0=1e-3,\n    augment=True,\n    workers=4,             # Use multiple cores for data loading\n    project=\"hubmap_project\",\n    name=\"run_v1\",\n    exist_ok=True\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run validation on the separate validation set\nmetrics = model.val()\nprint(f\"mAP50-95: {metrics.box.map}\")\nprint(f\"Seg mAP50-95: {metrics.seg.map}\")\n\n# Cleanup (Optional: deletes images to save space for commit)\n# shutil.rmtree(WORK_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}