{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"},{"sourceId":9245433,"sourceType":"datasetVersion","datasetId":5592926}],"dockerImageVersionId":31236,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics\nimport os\nimport pandas as pd\nimport cv2\nimport shutil\nfrom tqdm.notebook import tqdm\nfrom ultralytics import YOLO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T14:06:40.505071Z","iopub.execute_input":"2026-01-01T14:06:40.505339Z","iopub.status.idle":"2026-01-01T14:06:53.030860Z","shell.execute_reply.started":"2026-01-01T14:06:40.505316Z","shell.execute_reply":"2026-01-01T14:06:53.030054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CONFIGURATION ---\nimport os\nimport cv2\nimport csv\nimport random\nfrom collections import defaultdict\nfrom tqdm.notebook import tqdm\n\n# Install ultralytics (if not already installed after restart)\n# !pip install ultralytics \n\n# Kaggle Input Paths\nIMAGES_ROOT = '/kaggle/input/rsna-lumbar-spine-test-train-png-format/train_images' \nCOORD_CSV = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\n\n# Output Directory\nOUTPUT_DIR = '/kaggle/working/yolo_dataset'\n\n# Settings\nBOX_SIZE = 50 \nMAX_IMAGES = None  # Set to None for all images\n\n# Create Directories\nfor split in ['train', 'val']:\n    os.makedirs(f\"{OUTPUT_DIR}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{OUTPUT_DIR}/labels/{split}\", exist_ok=True)\n\n# --- PROCESSING WITHOUT PANDAS ---\nprint(\"Loading coordinates using standard CSV...\")\ngrouped_data = defaultdict(list)\n\n# Read CSV manually to avoid Pandas/NumPy conflict\nwith open(COORD_CSV, mode='r') as csvfile:\n    reader = csv.DictReader(csvfile)\n    for row in tqdm(reader, desc=\"Reading CSV\"):\n        # Create a unique key for grouping\n        key = (row['study_id'], row['series_id'], row['instance_number'])\n        grouped_data[key].append(row)\n\n# Convert dictionary to list for splitting\ngroups_list = list(grouped_data.items())\n\n# Shuffle and Split\nrandom.shuffle(groups_list)\nif MAX_IMAGES:\n    print(f\"Limiting to {MAX_IMAGES} images...\")\n    groups_list = groups_list[:MAX_IMAGES]\n\nsplit_idx = int(len(groups_list) * 0.8)\ntrain_groups = groups_list[:split_idx]\nval_groups = groups_list[split_idx:]\n\ndef process_batch(batch_groups, split_type='train'):\n    print(f\"Processing {split_type} data...\")\n    for (study_id, series_id, instance_num), rows in tqdm(batch_groups):\n        \n        img_rel_path = f\"{study_id}/{series_id}/{instance_num}.png\"\n        img_full_path = os.path.join(IMAGES_ROOT, img_rel_path)\n        \n        if not os.path.exists(img_full_path):\n            continue\n        \n        img = cv2.imread(img_full_path)\n        if img is None: continue\n        h, w = img.shape[:2]\n        \n        label_lines = []\n        for row in rows:\n            # Parse strings to floats manually\n            try:\n                x = float(row['x'])\n                y = float(row['y'])\n            except ValueError:\n                continue\n\n            # YOLO Format: class=0 x_center y_center width height (Normalized)\n            x_n = max(0, min(1, x / w))\n            y_n = max(0, min(1, y / h))\n            w_n = BOX_SIZE / w\n            h_n = BOX_SIZE / h\n            \n            label_lines.append(f\"0 {x_n:.6f} {y_n:.6f} {w_n:.6f} {h_n:.6f}\")\n            \n        # Save Label\n        file_id = f\"{study_id}_{series_id}_{instance_num}\"\n        with open(f\"{OUTPUT_DIR}/labels/{split_type}/{file_id}.txt\", 'w') as f:\n            f.write('\\n'.join(label_lines))\n            \n        # Save Image\n        cv2.imwrite(f\"{OUTPUT_DIR}/images/{split_type}/{file_id}.png\", img)\n\n# Run Processing\nprocess_batch(train_groups, 'train')\nprocess_batch(val_groups, 'val')\n\nprint(\"Dataset is ready in /kaggle/working/yolo_dataset\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T13:50:35.511786Z","iopub.execute_input":"2026-01-01T13:50:35.512189Z","iopub.status.idle":"2026-01-01T13:50:35.952216Z","shell.execute_reply.started":"2026-01-01T13:50:35.512155Z","shell.execute_reply":"2026-01-01T13:50:35.951178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_content = f\"\"\"\npath: {OUTPUT_DIR}\ntrain: images/train\nval: images/val\n\nnc: 1\nnames: ['Disc_Region']\n\"\"\"\n\nwith open('/kaggle/working/dataset.yaml', 'w') as f:\n    f.write(yaml_content)\n\nprint(\"dataset.yaml created successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T13:50:35.954191Z","iopub.execute_input":"2026-01-01T13:50:35.954503Z","iopub.status.idle":"2026-01-01T13:50:35.960258Z","shell.execute_reply.started":"2026-01-01T13:50:35.954440Z","shell.execute_reply":"2026-01-01T13:50:35.959439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = YOLO('yolov8n.pt')  # Load Nano model\n\nresults = model.train(\n    data='/kaggle/working/dataset.yaml',\n    epochs=15,          # 15 epochs is fine for a test\n    imgsz=512,\n    batch=16,\n    project='/kaggle/working/runs', # Save results to working dir\n    name='disc_detector',\n    exist_ok=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T13:50:35.961260Z","iopub.execute_input":"2026-01-01T13:50:35.961600Z","iopub.status.idle":"2026-01-01T13:50:47.044748Z","shell.execute_reply.started":"2026-01-01T13:50:35.961561Z","shell.execute_reply":"2026-01-01T13:50:47.043543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\n\nprint(\"--- DEBUGGING PATHS ---\")\n\n# 1. Check if Input Images exist\nINPUT_ROOT = '/kaggle/input/rsna-lumbar-spine-test-train-png-format/train_images'\nif os.path.exists(INPUT_ROOT):\n    print(f\"✅ Input folder found: {INPUT_ROOT}\")\n    # Print first few files to check structure\n    print(\"Sample input structure:\")\n    for root, dirs, files in os.walk(INPUT_ROOT):\n        if files:\n            print(f\"  Found files in: {root}\")\n            print(f\"  First file: {files[0]}\")\n            break\nelse:\n    print(f\"❌ Input folder NOT found at: {INPUT_ROOT}\")\n    print(\"Check your 'Add Data' panel to see the correct path.\")\n\n# 2. Check if Output Images exist\nOUTPUT_TRAIN = '/kaggle/working/yolo_dataset/images/train'\nif os.path.exists(OUTPUT_TRAIN):\n    num_files = len(os.listdir(OUTPUT_TRAIN))\n    print(f\"\\n📂 Output train folder exists. File count: {num_files}\")\n    if num_files == 0:\n        print(\"❌ ERROR: The train folder is EMPTY. The data prep script skipped all images.\")\nelse:\n    print(f\"\\n❌ Output folder does not exist: {OUTPUT_TRAIN}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T13:53:17.905703Z","iopub.execute_input":"2026-01-01T13:53:17.906087Z","iopub.status.idle":"2026-01-01T13:53:17.914053Z","shell.execute_reply.started":"2026-01-01T13:53:17.906057Z","shell.execute_reply":"2026-01-01T13:53:17.913212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport csv\nimport random\nfrom tqdm.notebook import tqdm\n\n# --- CONFIGURATION ---\n# Double check this path from the output of Step 1\nIMAGES_ROOT = '/kaggle/input/rsna-lumbar-spine-test-train-png-format/train_images' \nCOORD_CSV = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\nOUTPUT_DIR = '/kaggle/working/yolo_dataset'\nBOX_SIZE = 50 \nMAX_IMAGES = 500  # Keep it small (500) for this test run!\n\n# Clean up old empty folders\nimport shutil\nif os.path.exists(OUTPUT_DIR):\n    shutil.rmtree(OUTPUT_DIR)\n\n# Create Directories\nfor split in ['train', 'val']:\n    os.makedirs(f\"{OUTPUT_DIR}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{OUTPUT_DIR}/labels/{split}\", exist_ok=True)\n\n# --- HELPER: Find Image Path (Handles '1' vs '001') ---\ndef find_image_path(root, study, series, instance):\n    # Try direct path\n    path = os.path.join(root, str(study), str(series), f\"{instance}.png\")\n    if os.path.exists(path): return path\n    \n    # Try with leading zeros if needed? (Usually RSNA is clean, but let's be safe)\n    # Most common issue: The dataset structure is just distinct.\n    # Let's assume standard structure first.\n    return None\n\n# --- PROCESSING ---\nprint(\"Reading CSV and grouping data...\")\ngrouped_data = {}\n\nwith open(COORD_CSV, mode='r') as f:\n    reader = csv.DictReader(f)\n    for row in tqdm(reader, desc=\"Parsing CSV\"):\n        # Key: study/series/instance\n        k = (row['study_id'], row['series_id'], row['instance_number'])\n        if k not in grouped_data: grouped_data[k] = []\n        grouped_data[k].append(row)\n\nkeys = list(grouped_data.keys())\nrandom.shuffle(keys)\nif MAX_IMAGES: keys = keys[:MAX_IMAGES]\n\n# Split\nsplit_idx = int(len(keys) * 0.8)\ntrain_keys = keys[:split_idx]\nval_keys = keys[split_idx:]\n\nsuccess_count = 0\n\ndef process(keys, split):\n    global success_count\n    for (study, series, instance) in tqdm(keys, desc=f\"Processing {split}\"):\n        \n        # Construct Path\n        img_path = os.path.join(IMAGES_ROOT, str(study), str(series), f\"{instance}.png\")\n        \n        if not os.path.exists(img_path):\n            # Debug: Print first failure only\n            # print(f\"Missing: {img_path}\") \n            continue\n            \n        img = cv2.imread(img_path)\n        if img is None: continue\n        h, w = img.shape[:2]\n        \n        label_lines = []\n        rows = grouped_data[(study, series, instance)]\n        \n        for row in rows:\n            try:\n                x, y = float(row['x']), float(row['y'])\n                x_n, y_n = x/w, y/h\n                w_n, h_n = BOX_SIZE/w, BOX_SIZE/h\n                label_lines.append(f\"0 {x_n:.6f} {y_n:.6f} {w_n:.6f} {h_n:.6f}\")\n            except: continue\n            \n        # Save\n        fname = f\"{study}_{series}_{instance}\"\n        with open(f\"{OUTPUT_DIR}/labels/{split}/{fname}.txt\", 'w') as f:\n            f.write('\\n'.join(label_lines))\n        cv2.imwrite(f\"{OUTPUT_DIR}/images/{split}/{fname}.png\", img)\n        success_count += 1\n\nprocess(train_keys, 'train')\nprocess(val_keys, 'val')\n\nprint(f\"\\n✅ Done! Successfully saved {success_count} images.\")\nif success_count == 0:\n    print(\"❌ CRITICAL: No images were saved. Check IMAGES_ROOT path again!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T13:53:57.743314Z","iopub.execute_input":"2026-01-01T13:53:57.744233Z","iopub.status.idle":"2026-01-01T13:53:58.347990Z","shell.execute_reply.started":"2026-01-01T13:53:57.744185Z","shell.execute_reply":"2026-01-01T13:53:58.346921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"--- SEARCHING FOR IMAGES ---\")\n# Walk through the input directory to find where the .png files really are\nfound = False\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file.endswith('.png'):\n            print(f\"✅ FOUND IMAGES HERE: {root}\")\n            print(f\"   Example file: {file}\")\n            found = True\n            break\n    if found: break\n\nif not found:\n    print(\"❌ No PNG images found. Did you add the dataset?\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T14:04:52.335206Z","iopub.execute_input":"2026-01-01T14:04:52.335452Z","iopub.status.idle":"2026-01-01T14:05:40.373961Z","shell.execute_reply.started":"2026-01-01T14:04:52.335430Z","shell.execute_reply":"2026-01-01T14:05:40.372860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport csv\nimport random\nfrom tqdm.notebook import tqdm\nimport shutil\n\n# --- CONFIGURATION (UPDATED) ---\n# The path has been fixed to match your screenshot\nIMAGES_ROOT = '/kaggle/input/rsna-lumbar-spine-test-train-png-format/train_images_png'\n\nCOORD_CSV = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\nOUTPUT_DIR = '/kaggle/working/yolo_dataset'\nBOX_SIZE = 50 \nMAX_IMAGES = None  # Set to None later for the full dataset\n\n# --- SAFETY CHECK ---\nif not os.path.exists(IMAGES_ROOT):\n    print(f\"❌ STOP! The path still does not exist: {IMAGES_ROOT}\")\nelse:\n    print(f\"✅ Path found: {IMAGES_ROOT}\")\n\n    # Clean and Create Dirs\n    if os.path.exists(OUTPUT_DIR): \n        shutil.rmtree(OUTPUT_DIR)\n        \n    for split in ['train', 'val']:\n        os.makedirs(f\"{OUTPUT_DIR}/images/{split}\", exist_ok=True)\n        os.makedirs(f\"{OUTPUT_DIR}/labels/{split}\", exist_ok=True)\n\n    # --- PROCESSING ---\n    print(\"Reading CSV...\")\n    grouped_data = {}\n    \n    # Read CSV manually\n    with open(COORD_CSV, mode='r') as f:\n        reader = csv.DictReader(f)\n        for row in tqdm(reader, desc=\"Parsing CSV\"):\n            # Create key: (study_id, series_id, instance_number)\n            k = (row['study_id'], row['series_id'], row['instance_number'])\n            if k not in grouped_data: grouped_data[k] = []\n            grouped_data[k].append(row)\n\n    # Select random sample\n    keys = list(grouped_data.keys())\n    random.shuffle(keys)\n    if MAX_IMAGES: \n        keys = keys[:MAX_IMAGES]\n\n    # Split Train/Val\n    split_idx = int(len(keys) * 0.8)\n    train_keys = keys[:split_idx]\n    val_keys = keys[split_idx:]\n\n    success_count = 0\n\n    def process(keys, split):\n        global success_count\n        for (study, series, instance) in tqdm(keys, desc=f\"Processing {split}\"):\n            \n            # Construct Path: Root / Study / Series / Instance.png\n            img_path = os.path.join(IMAGES_ROOT, str(study), str(series), f\"{instance}.png\")\n            \n            if not os.path.exists(img_path):\n                continue\n                \n            img = cv2.imread(img_path)\n            if img is None: continue\n            h, w = img.shape[:2]\n            \n            label_lines = []\n            rows = grouped_data[(study, series, instance)]\n            \n            for row in rows:\n                try:\n                    # Parse coords\n                    x, y = float(row['x']), float(row['y'])\n                    \n                    # Normalize for YOLO (0-1 range)\n                    x_n, y_n = x/w, y/h\n                    w_n, h_n = BOX_SIZE/w, BOX_SIZE/h\n                    \n                    # Ensure valid range\n                    x_n = max(0, min(1, x_n))\n                    y_n = max(0, min(1, y_n))\n                    \n                    label_lines.append(f\"0 {x_n:.6f} {y_n:.6f} {w_n:.6f} {h_n:.6f}\")\n                except: \n                    continue\n            \n            # Save Labels and Image\n            if label_lines:\n                fname = f\"{study}_{series}_{instance}\"\n                \n                # Write .txt file\n                with open(f\"{OUTPUT_DIR}/labels/{split}/{fname}.txt\", 'w') as f:\n                    f.write('\\n'.join(label_lines))\n                \n                # Save .png file\n                cv2.imwrite(f\"{OUTPUT_DIR}/images/{split}/{fname}.png\", img)\n                success_count += 1\n\n    process(train_keys, 'train')\n    process(val_keys, 'val')\n\n    print(f\"\\n✅ Success! Saved {success_count} images.\")\n    print(\"You can now run the Training Cell.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T14:14:53.847766Z","iopub.execute_input":"2026-01-01T14:14:53.848581Z","iopub.status.idle":"2026-01-01T14:23:12.893790Z","shell.execute_reply.started":"2026-01-01T14:14:53.848547Z","shell.execute_reply":"2026-01-01T14:23:12.893083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the YAML file that points to your new dataset\nyaml_content = f\"\"\"\npath: /kaggle/working/yolo_dataset\ntrain: images/train\nval: images/val\n\nnc: 1\nnames: ['Disc_Region']\n\"\"\"\n\nwith open('/kaggle/working/dataset.yaml', 'w') as f:\n    f.write(yaml_content)\n\nprint(\"✅ dataset.yaml created successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T14:24:23.891846Z","iopub.execute_input":"2026-01-01T14:24:23.892621Z","iopub.status.idle":"2026-01-01T14:24:23.897586Z","shell.execute_reply.started":"2026-01-01T14:24:23.892591Z","shell.execute_reply":"2026-01-01T14:24:23.896981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load the model\nmodel = YOLO('yolov8n.pt') \n\n# Train\nprint(\"Starting training...\")\nresults = model.train(\n    data='/kaggle/working/dataset.yaml',\n    epochs=15,\n    imgsz=512,\n    batch=16,\n    project='/kaggle/working/runs',\n    name='disc_detector',\n    exist_ok=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T14:24:40.250933Z","iopub.execute_input":"2026-01-01T14:24:40.251263Z","iopub.status.idle":"2026-01-01T15:15:25.431073Z","shell.execute_reply.started":"2026-01-01T14:24:40.251233Z","shell.execute_reply":"2026-01-01T15:15:25.430350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport glob\nimport cv2\n\n# Path to your trained model weights\nweights_path = '/kaggle/working/runs/disc_detector/weights/best.pt'\n\nif os.path.exists(weights_path):\n    best_model = YOLO(weights_path)\n    \n    # Pick a random image from validation set\n    val_images = glob.glob(\"/kaggle/working/yolo_dataset/images/val/*.png\")\n    \n    if val_images:\n        # Predict on the first image found\n        test_img = val_images[0]\n        results = best_model.predict(test_img)\n        \n        # Plot result\n        plt.figure(figsize=(10, 10))\n        # results[0].plot() returns a BGR numpy array, convert to RGB for matplotlib\n        plt.imshow(cv2.cvtColor(results[0].plot(), cv2.COLOR_BGR2RGB))\n        plt.axis('off')\n        plt.title(\"Detected Disc Regions\")\n        plt.show()\n    else:\n        print(\"No validation images found to test.\")\nelse:\n    print(f\"Model file not found at {weights_path}. Did training finish?\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T15:19:24.295882Z","iopub.execute_input":"2026-01-01T15:19:24.296559Z","iopub.status.idle":"2026-01-01T15:19:24.629346Z","shell.execute_reply.started":"2026-01-01T15:19:24.296511Z","shell.execute_reply":"2026-01-01T15:19:24.628639Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"preprocessing","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport csv\nimport random\nimport numpy as np\nimport shutil\nfrom tqdm.notebook import tqdm\n\n# --- CONFIGURATION ---\nIMAGES_ROOT = '/kaggle/input/rsna-lumbar-spine-test-train-png-format/train_images_png'\nCOORD_CSV = '/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train_label_coordinates.csv'\nOUTPUT_DIR = '/kaggle/working/yolo_dataset'\nBOX_SIZE = 50 \nMAX_IMAGES = None  # Set to None for the full dataset\n\n# --- ADVANCED PREPROCESSING FUNCTION ---\ndef preprocess_image(img):\n    # 1. Convert to Grayscale\n    if len(img.shape) == 3:\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    \n    # 2. Intensity Rescaling (Min-Max Scaling to 0-255)\n    # This ensures the image uses the full dynamic range before CLAHE\n    img_float = img.astype(float)\n    img_rescaled = 255 * (img_float - img_float.min()) / (img_float.max() - img_float.min() + 1e-8)\n    img = img_rescaled.astype(np.uint8)\n\n    # 3. Apply CLAHE (Contrast Enhancement)\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8))\n    img = clahe.apply(img)\n    \n    # 4. Normalize (Z-score normalization simulation)\n    # YOLO typically handles normalization internally during training (0-1 scaling),\n    # but stabilizing the input histogram here helps convergence.\n    # We will normalize and then clip back to 8-bit for saving as PNG.\n    img_norm = img.astype(float)\n    mean, std = img_norm.mean(), img_norm.std()\n    img_norm = (img_norm - mean) / (std + 1e-8)\n    \n    # Scale back to 0-255 visually for PNG saving\n    # (We map -3 std to +3 std range into 0-255)\n    img_norm = np.clip(((img_norm * 40) + 128), 0, 255)\n    final_img = img_norm.astype(np.uint8)\n    \n    return final_img\n\n# --- CLEAR OLD DATASET ---\nif os.path.exists(OUTPUT_DIR):\n    shutil.rmtree(OUTPUT_DIR)\n    print(\"Old dataset deleted.\")\n\n# --- CREATE DIRECTORIES ---\nfor split in ['train', 'val']:\n    os.makedirs(f\"{OUTPUT_DIR}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{OUTPUT_DIR}/labels/{split}\", exist_ok=True)\n\n# --- READ LABELS ---\nprint(\"Reading CSV...\")\ngrouped_data = {}\nwith open(COORD_CSV, mode='r') as f:\n    reader = csv.DictReader(f)\n    for row in tqdm(reader, desc=\"Parsing CSV\"):\n        k = (row['study_id'], row['series_id'], row['instance_number'])\n        if k not in grouped_data: grouped_data[k] = []\n        grouped_data[k].append(row)\n\nkeys = list(grouped_data.keys())\nrandom.shuffle(keys)\nif MAX_IMAGES: keys = keys[:MAX_IMAGES]\n\n# Split 80/20\nsplit_idx = int(len(keys) * 0.8)\ntrain_keys = keys[:split_idx]\nval_keys = keys[split_idx:]\n\nsuccess_count = 0\n\ndef process_and_save(keys, split):\n    global success_count\n    for (study, series, instance) in tqdm(keys, desc=f\"Preprocessing {split}\"):\n        \n        img_path = os.path.join(IMAGES_ROOT, str(study), str(series), f\"{instance}.png\")\n        if not os.path.exists(img_path): continue\n            \n        img = cv2.imread(img_path)\n        if img is None: continue\n        h, w = img.shape[:2]\n        \n        # --- APPLY FULL PIPELINE ---\n        final_img = preprocess_image(img)\n        \n        # Create Labels\n        label_lines = []\n        rows = grouped_data[(study, series, instance)]\n        \n        for row in rows:\n            try:\n                x, y = float(row['x']), float(row['y'])\n                x_n, y_n = x/w, y/h\n                w_n, h_n = BOX_SIZE/w, BOX_SIZE/h\n                x_n = max(0, min(1, x_n))\n                y_n = max(0, min(1, y_n))\n                label_lines.append(f\"0 {x_n:.6f} {y_n:.6f} {w_n:.6f} {h_n:.6f}\")\n            except: continue\n            \n        if label_lines:\n            fname = f\"{study}_{series}_{instance}\"\n            with open(f\"{OUTPUT_DIR}/labels/{split}/{fname}.txt\", 'w') as f:\n                f.write('\\n'.join(label_lines))\n            cv2.imwrite(f\"{OUTPUT_DIR}/images/{split}/{fname}.png\", final_img)\n            success_count += 1\n\nprocess_and_save(train_keys, 'train')\nprocess_and_save(val_keys, 'val')\n\nprint(f\"\\n✅ Full Preprocessing Complete! {success_count} images ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T15:43:44.525862Z","iopub.execute_input":"2026-01-01T15:43:44.526239Z","iopub.status.idle":"2026-01-01T15:49:39.680802Z","shell.execute_reply.started":"2026-01-01T15:43:44.526208Z","shell.execute_reply":"2026-01-01T15:49:39.679987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the YAML file\nyaml_content = f\"\"\"\npath: /kaggle/working/yolo_dataset\ntrain: images/train\nval: images/val\n\nnc: 1\nnames: ['Disc_Region']\n\"\"\"\n\nwith open('/kaggle/working/dataset.yaml', 'w') as f:\n    f.write(yaml_content)\n\nprint(\"✅ dataset.yaml ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T15:51:20.398462Z","iopub.execute_input":"2026-01-01T15:51:20.399120Z","iopub.status.idle":"2026-01-01T15:51:20.403897Z","shell.execute_reply.started":"2026-01-01T15:51:20.399083Z","shell.execute_reply":"2026-01-01T15:51:20.403268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load the nano model\nmodel = YOLO('yolov8n.pt') \n\n# Start Training\nprint(\"Starting training on preprocessed data...\")\nresults = model.train(\n    data='/kaggle/working/dataset.yaml',\n    epochs=15, \n    imgsz=512,\n    batch=16,\n    project='/kaggle/working/runs',\n    name='disc_detector_preprocessed', # New name to distinguish from previous run\n    exist_ok=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-01T15:51:59.932980Z","iopub.execute_input":"2026-01-01T15:51:59.933307Z","iopub.status.idle":"2026-01-01T16:41:40.905941Z","shell.execute_reply.started":"2026-01-01T15:51:59.933280Z","shell.execute_reply":"2026-01-01T16:41:40.905221Z"}},"outputs":[],"execution_count":null}]}