{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":71549,"databundleVersionId":8561470}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Preprocessing - Convert RSNA to YOLO\nimport pandas as pd\nimport numpy as np\nimport pydicom\nimport cv2\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nimport os\n\n# =====================\n# Paths & Config\n# =====================\ninput_base_path = Path('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification')\noutput_base_path = Path('/kaggle/working/yolo_dataset')\n\nclass_map = {\n    \"Spinal Canal Stenosis\": 0,\n    \"Right Neural Foraminal Narrowing\": 1,\n    \"Left Neural Foraminal Narrowing\": 2,\n    \"Left Subarticular Stenosis\": 3,\n    \"Right Subarticular Stenosis\": 4,\n}\n\nBOX_WIDTH_RATIO = 0.15\nBOX_HEIGHT_RATIO = 0.2\nTARGET_SIZE = 640\n\n# =====================\n# Helper: Resize with Letterbox\n# =====================\ndef resize_and_letterbox(img, new_size=640):\n    h, w = img.shape[:2]\n    scale = min(new_size / w, new_size / h)\n    new_w, new_h = int(w * scale), int(h * scale)\n\n    # Resize keeping aspect ratio\n    resized = cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_AREA)\n\n    # Create square canvas\n    canvas = np.full((new_size, new_size, 3), 0, dtype=np.uint8)\n    top = (new_size - new_h) // 2\n    left = (new_size - new_w) // 2\n    canvas[top:top+new_h, left:left+new_w] = resized\n\n    return canvas, scale, left, top\n\n# =====================\n# Load Data\n# =====================\nprint(\"Loading dataframes...\")\ndf_coords = pd.read_csv(input_base_path / 'train_label_coordinates.csv')\ndf_series_desc = pd.read_csv(input_base_path / 'train_series_descriptions.csv')\n\nprint(f\"Coords shape: {df_coords.shape}\")\nprint(f\"Series desc shape: {df_series_desc.shape}\")\n\ndf_coords_filtered = df_coords[df_coords['condition'].isin(class_map.keys())]\nprint(f\"After filtering: {df_coords_filtered.shape}\")\n\nif len(df_coords_filtered) > 0:\n    all_studies = df_coords_filtered['study_id'].unique()\n    print(f\"Unique studies: {len(all_studies)}\")\n\n    if len(all_studies) > 2:\n\n        train_studies, temp_studies = train_test_split(\n            all_studies, test_size=0.30, random_state=42\n        )\n        val_studies, test_studies = train_test_split(\n            temp_studies, test_size=0.50, random_state=42\n        )\n\n        train_df = df_coords_filtered[df_coords_filtered['study_id'].isin(train_studies)]\n        val_df   = df_coords_filtered[df_coords_filtered['study_id'].isin(val_studies)]\n        test_df  = df_coords_filtered[df_coords_filtered['study_id'].isin(test_studies)]\n\n        print(f\"Train labels: {len(train_df)} | Val: {len(val_df)} | Test: {len(test_df)}\")\n\n        # =====================\n        # Main Dataset Creator\n        # =====================\n        def create_yolo_dataset(dataframe, dataset_type='train'):\n            image_output_dir = output_base_path / dataset_type / 'images'\n            label_output_dir = output_base_path / dataset_type / 'labels'\n            image_output_dir.mkdir(parents=True, exist_ok=True)\n            label_output_dir.mkdir(parents=True, exist_ok=True)\n\n            processed_count = 0\n            grouped = dataframe.groupby(['study_id', 'series_id', 'instance_number'])\n\n            for (study_id, series_id, instance_number), group in grouped:\n                study_id_str = str(study_id)\n                series_id_str = str(series_id)\n                instance_number_str = str(instance_number)\n\n                dcm_path = input_base_path / 'train_images' / study_id_str / series_id_str / f\"{instance_number_str}.dcm\"\n                jpg_filename = f\"{study_id_str}{series_id_str}{instance_number_str}.jpg\"\n                txt_filename = f\"{study_id_str}{series_id_str}{instance_number_str}.txt\"\n\n                label_file_path = label_output_dir / txt_filename\n\n                if not dcm_path.exists():\n                    continue\n\n                try:\n                    ds = pydicom.dcmread(dcm_path)\n                    img_pixels = ds.pixel_array\n\n                    # DICOM windowing\n                    if hasattr(ds, 'WindowCenter') and hasattr(ds, 'WindowWidth'):\n                        center = ds.WindowCenter[0] if isinstance(ds.WindowCenter, pydicom.multival.MultiValue) else ds.WindowCenter\n                        width = ds.WindowWidth[0] if isinstance(ds.WindowWidth, pydicom.multival.MultiValue) else ds.WindowWidth\n                        low, high = center - width / 2, center + width / 2\n                        if high > low:\n                            img_pixels = np.clip(img_pixels, low, high)\n                            img_pixels = ((img_pixels - low) / (high - low) * 255).astype(np.uint8)\n                    else:\n                        img_pixels = ((img_pixels - img_pixels.min()) / (img_pixels.max() - img_pixels.min()) * 255).astype(np.uint8)\n\n                    # Convert grayscale → RGB\n                    if len(img_pixels.shape) == 2:\n                        img_pixels = cv2.cvtColor(img_pixels, cv2.COLOR_GRAY2RGB)\n\n                    # Resize to 640x640 with letterboxing\n                    resized_img, scale, left, top = resize_and_letterbox(img_pixels, new_size=TARGET_SIZE)\n\n                    # Save resized image\n                    jpg_path = image_output_dir / jpg_filename\n                    if not jpg_path.exists():\n                        cv2.imwrite(str(jpg_path), resized_img)\n\n                    img_height, img_width = TARGET_SIZE, TARGET_SIZE\n\n                except Exception as e:\n                    print(f\"Error {dcm_path}: {e}\")\n                    continue\n\n                # Collect YOLO labels\n                lines = []\n                for _, row in group.iterrows():\n                    if pd.isna(row['x']) or pd.isna(row['y']):\n                        continue\n\n                    # Original coords\n                    x_center, y_center = row['x'], row['y']\n\n                    # Apply scaling + padding\n                    x_center = x_center * scale + left\n                    y_center = y_center * scale + top\n\n                    # Fixed box size (scaled)\n                    w_pixels = BOX_WIDTH_RATIO * (img_width / scale) * scale\n                    h_pixels = BOX_HEIGHT_RATIO * (img_height / scale) * scale\n\n                    # Normalize to YOLO format\n                    x_norm = x_center / img_width\n                    y_norm = y_center / img_height\n                    w_norm = w_pixels / img_width\n                    h_norm = h_pixels / img_height\n\n                    class_id = class_map.get(row['condition'])\n                    if class_id is not None:\n                        lines.append(f\"{class_id} {x_norm:.6f} {y_norm:.6f} {w_norm:.6f} {h_norm:.6f}\\n\")\n\n                # Write label file\n                with open(label_file_path, 'w') as f:\n                    f.writelines(lines)\n\n                processed_count += 1\n                if processed_count % 1000 == 0:\n                    print(f\"Processed {processed_count} images for {dataset_type}\")\n\n            print(f\"Finished {dataset_type} set: {processed_count} images\")\n\n        # Run preprocessing\n        create_yolo_dataset(train_df, 'train')\n        create_yolo_dataset(val_df, 'val')\n        create_yolo_dataset(test_df, 'test')\n\n        # =====================\n        # Create dataset.yaml\n        # =====================\n        names_str = \"\\n\".join([f\"  - {n}\" for n in class_map.keys()])\n        yaml_content = f\"\"\"\npath: {output_base_path}\ntrain: train/images\nval: val/images\ntest: test/images\n\nnc: {len(class_map)}\n\nnames:\n{names_str}\n\"\"\"\n        yaml_file_path = output_base_path / \"dataset.yaml\"\n        with open(yaml_file_path, 'w') as f:\n            f.write(yaml_content.strip())\n\n        print(\"dataset.yaml created\")\n        print(yaml_content)\n\n    else:\n        print(\"Not enough studies for split\")\nelse:\n    print(\"No data after filtering\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T14:26:49.026486Z","iopub.execute_input":"2025-09-17T14:26:49.026651Z","iopub.status.idle":"2025-09-17T14:35:24.300118Z","shell.execute_reply.started":"2025-09-17T14:26:49.026636Z","shell.execute_reply":"2025-09-17T14:35:24.299264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q ultralytics==8.3.27 --no-deps\n!pip install -q \"ray[tune]\"==2.9.3  \n\nfrom ultralytics import YOLO\n\n# Load YOLOv10 \nmodel = YOLO('yolov10n.pt')   # yolov10s.pt, yolov10m.pt \n\nresults = model.train(\n    data=\"/kaggle/working/yolo_dataset/dataset.yaml\",\n    epochs=50,\n    imgsz=640,\n    batch=16,\n    workers=2,\n    project=\"rsna_sciatica_detection\",\n    name=\"yolov10_rsna\",\n    exist_ok=True,\n    save=True,\n    save_period=5   # save weights every 5 epochs\n)\n\n\n# Evaluate on test set\nmetrics = model.val(\n    data=\"/kaggle/working/yolo_dataset/dataset.yaml\",\n    split=\"test\"\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T14:48:31.418034Z","iopub.execute_input":"2025-09-17T14:48:31.418734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install ultralytics with YOLO11 support\n!pip install -q ultralytics --upgrade\n\nfrom ultralytics import YOLO\n\n# Load YOLOv11 \nmodel = YOLO('yolo11n.pt')   # yolo11s.pt, yolo11m.pt, yolo11l.pt, yolo11x.pt\n\n# Train the model\nresults = model.train(\n    data=\"/kaggle/working/yolo_dataset/dataset.yaml\",\n    epochs=50,\n    imgsz=640,\n    batch=16,\n    workers=2,\n    project=\"rsna_sciatica_detection\",\n    name=\"yolo11_rsna\",\n    exist_ok=True,\n    save=True,\n    save_period=5   # save weights every 5 epochs\n)\n\n# Evaluate on test set\nmetrics = model.val(\n    data=\"/kaggle/working/yolo_dataset/dataset.yaml\",\n    split=\"test\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T14:44:48.618544Z","iopub.execute_input":"2025-09-17T14:44:48.619215Z","iopub.status.idle":"2025-09-17T14:47:09.477732Z","shell.execute_reply.started":"2025-09-17T14:44:48.619185Z","shell.execute_reply":"2025-09-17T14:47:09.47666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nimport cv2\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nimport os\n\n# 1. Define Configuration\ninput_base_path = Path('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification')\noutput_base_path = Path('/kaggle/working/yolo_dataset')\n\n# Define your class mapping based on ACTUAL condition names in your dataset\nclass_map = {\n    \"Spinal Canal Stenosis\": 0,\n    \"Right Neural Foraminal Narrowing\": 1,\n    \"Left Neural Foraminal Narrowing\": 2,\n    \"Left Subarticular Stenosis\": 3,\n    \"Right Subarticular Stenosis\": 4,\n}\n\n# Define box size ratios\nBOX_WIDTH_RATIO = 0.15\nBOX_HEIGHT_RATIO = 0.2\n\n# 2. Load and Prepare DataFrames\nprint(\"Loading dataframes...\")\ndf_coords = pd.read_csv(input_base_path / 'train_label_coordinates.csv')\ndf_series_desc = pd.read_csv(input_base_path / 'train_series_descriptions.csv')\n\nprint(f\"Original coordinates dataframe shape: {df_coords.shape}\")\nprint(f\"Original series descriptions dataframe shape: {df_series_desc.shape}\")\n\n# 3. Filter for relevant conditions\nprint(\"\\nFiltering for selected conditions...\")\ndf_coords_filtered = df_coords[df_coords['condition'].isin(class_map.keys())]\nprint(f\"After condition filtering: {df_coords_filtered.shape}\")\n\n# 4. If we have data, proceed with the split\nif len(df_coords_filtered) > 0:\n    print(\"Creating train/validation/test split...\")\n    all_studies = df_coords_filtered['study_id'].unique()\n    print(f\"Number of unique studies: {len(all_studies)}\")\n    \n    if len(all_studies) > 2:\n        # Step 1: Train (70%) vs Temp (30%)\n        train_studies, temp_studies = train_test_split(\n            all_studies, test_size=0.30, random_state=42\n        )\n        \n        # Step 2: Split Temp into Val (15%) and Test (15%)\n        val_studies, test_studies = train_test_split(\n            temp_studies, test_size=0.50, random_state=42\n        )\n        \n        # Create DataFrames\n        train_df = df_coords_filtered[df_coords_filtered['study_id'].isin(train_studies)]\n        val_df   = df_coords_filtered[df_coords_filtered['study_id'].isin(val_studies)]\n        test_df  = df_coords_filtered[df_coords_filtered['study_id'].isin(test_studies)]\n\n        print(f\"Total label rows to process: {len(df_coords_filtered)}\")\n        print(f\"Training set labels: {len(train_df)}\")\n        print(f\"Validation set labels: {len(val_df)}\")\n        print(f\"Test set labels: {len(test_df)}\")\n\n        # Count unique images\n        train_unique_images = train_df.groupby(['study_id', 'series_id', 'instance_number']).ngroups\n        val_unique_images = val_df.groupby(['study_id', 'series_id', 'instance_number']).ngroups\n        test_unique_images = test_df.groupby(['study_id', 'series_id', 'instance_number']).ngroups\n        print(f\"Unique training images: {train_unique_images}\")\n        print(f\"Unique validation images: {val_unique_images}\")\n        print(f\"Unique test images: {test_unique_images}\")\n        \n        # 5. Define the Processing Function\n        def create_yolo_dataset(dataframe, dataset_type='train'):\n            \"\"\"\n            Processes a dataframe and creates YOLO images and labels in the output directory.\n            \"\"\"\n            image_output_dir = output_base_path / dataset_type / 'images'\n            label_output_dir = output_base_path / dataset_type / 'labels'\n            image_output_dir.mkdir(parents=True, exist_ok=True)\n            label_output_dir.mkdir(parents=True, exist_ok=True)\n            \n            processed_count = 0\n            # Group by (study, series, instance) to avoid reading the same DICOM file multiple times\n            grouped = dataframe.groupby(['study_id', 'series_id', 'instance_number'])\n            \n            for (study_id, series_id, instance_number), group in grouped:\n                study_id_str = str(study_id)\n                series_id_str = str(series_id)\n                instance_number_str = str(instance_number)\n                \n                dcm_path = input_base_path / 'train_images' / study_id_str / series_id_str / f\"{instance_number_str}.dcm\"\n                jpg_filename = f\"{study_id_str}_{series_id_str}_{instance_number_str}.jpg\"\n                txt_filename = f\"{study_id_str}_{series_id_str}_{instance_number_str}.txt\"\n                \n                label_file_path = label_output_dir / txt_filename\n                \n                if not dcm_path.exists():\n                    continue\n                    \n                try:\n                    ds = pydicom.dcmread(dcm_path)\n                    img_pixels = ds.pixel_array\n                    \n                    if hasattr(ds, 'WindowCenter') and hasattr(ds, 'WindowWidth'):\n                        if isinstance(ds.WindowCenter, pydicom.multival.MultiValue):\n                            center = ds.WindowCenter[0]\n                            width = ds.WindowWidth[0]\n                        else:\n                            center = ds.WindowCenter\n                            width = ds.WindowWidth\n                        \n                        low = center - width / 2\n                        high = center + width / 2\n                        \n                        img_pixels = np.clip(img_pixels, low, high)\n                        img_pixels = ((img_pixels - low) / (high - low) * 255).astype(np.uint8)\n                    else:\n                        img_pixels = ((img_pixels - img_pixels.min()) / (img_pixels.max() - img_pixels.min()) * 255).astype(np.uint8)\n                    \n                    jpg_path = image_output_dir / jpg_filename\n                    if not jpg_path.exists():\n                        cv2.imwrite(str(jpg_path), img_pixels)\n                    \n                    img_height, img_width = img_pixels.shape\n                    \n                except Exception as e:\n                    print(f\"Error processing {dcm_path}: {e}\")\n                    continue\n                \n                for _, row in group.iterrows():\n                    condition = row['condition']\n                    x_center = row['x']\n                    y_center = row['y']\n                    \n                    if pd.isna(x_center) or pd.isna(y_center):\n                        continue\n                        \n                    # Calculate bounding box in pixels\n                    w_pixels = BOX_WIDTH_RATIO * img_width\n                    h_pixels = BOX_HEIGHT_RATIO * img_height\n                    \n                    # Normalize to YOLO format\n                    x_center_norm = x_center / img_width\n                    y_center_norm = y_center / img_height\n                    w_norm = w_pixels / img_width\n                    h_norm = h_pixels / img_height\n                    \n                    class_id = class_map.get(condition)\n                    if class_id is None:\n                        continue\n                        \n                    yolo_line = f\"{class_id} {x_center_norm:.6f} {y_center_norm:.6f} {w_norm:.6f} {h_norm:.6f}\\n\"\n                    \n                    with open(label_file_path, 'a') as f:\n                        f.write(yolo_line)\n                \n                processed_count += 1\n                if processed_count % 1000 == 0:\n                    print(f\"Processed {processed_count} images for {dataset_type} set...\")\n            \n            print(f\"Finished processing {dataset_type} set. Total images: {processed_count}\")\n        \n        # 6. Run the processing for all sets\n        print(\"Starting processing for training set...\")\n        create_yolo_dataset(train_df, 'train')\n\n        print(\"Starting processing for validation set...\")\n        create_yolo_dataset(val_df, 'val')\n\n        print(\"Starting processing for test set...\")\n        create_yolo_dataset(test_df, 'test')\n\n        print(\"Dataset creation complete!\")\n\n        # 7. Create the dataset.yaml file\n        yaml_content = f\"\"\"\npath: {output_base_path}\ntrain: train/images\nval: val/images\ntest: test/images\n\n# Number of classes\nnc: {len(class_map)}\n\n# Class names\nnames: {list(class_map.keys())}\n\"\"\"\n\n        yaml_file_path = output_base_path / \"dataset.yaml\"\n        with open(yaml_file_path, 'w') as f:\n            f.write(yaml_content.strip())\n            \n        print(\"dataset.yaml created successfully!\")\n        print(f\"YAML file content:\\n{yaml_content}\")\n\n        # 8. Verify the output\n        print(\"\\nVerifying output structure...\")\n        train_images = len(list((output_base_path / 'train' / 'images').glob('*.jpg')))\n        train_labels = len(list((output_base_path / 'train' / 'labels').glob('*.txt')))\n        val_images = len(list((output_base_path / 'val' / 'images').glob('*.jpg')))\n        val_labels = len(list((output_base_path / 'val' / 'labels').glob('*.txt')))\n        test_images = len(list((output_base_path / 'test' / 'images').glob('*.jpg')))\n        test_labels = len(list((output_base_path / 'test' / 'labels').glob('*.txt')))\n\n        print(f\"Training images: {train_images}\")\n        print(f\"Training labels: {train_labels}\")\n        print(f\"Validation images: {val_images}\")\n        print(f\"Validation labels: {val_labels}\")\n        print(f\"Test images: {test_images}\")\n        print(f\"Test labels: {test_labels}\")\n\n    else:\n        print(\"Not enough studies for train/validation/test split\")\nelse:\n    print(\"No data available after filtering. Please check your filtering conditions.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2025-09-10T07:47:23.358568Z","iopub.execute_input":"2025-09-10T07:47:23.358862Z","iopub.status.idle":"2025-09-10T07:55:37.826083Z","shell.execute_reply.started":"2025-09-10T07:47:23.35884Z","shell.execute_reply":"2025-09-10T07:55:37.825297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing - Convert RSNA to YOLO\nimport pandas as pd\nimport numpy as np\nimport pydicom\nimport cv2\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nimport os\n\n# =====================\n# Paths & Config\n# =====================\ninput_base_path = Path('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification')\noutput_base_path = Path('/kaggle/working/yolo_dataset')\n\nclass_map = {\n    \"Spinal Canal Stenosis\": 0,\n    \"Right Neural Foraminal Narrowing\": 1,\n    \"Left Neural Foraminal Narrowing\": 2,\n    \"Left Subarticular Stenosis\": 3,\n    \"Right Subarticular Stenosis\": 4,\n}\n\nBOX_WIDTH_RATIO = 0.15\nBOX_HEIGHT_RATIO = 0.2\nTARGET_SIZE = 640\n\n# =====================\n# Helper: Resize with Letterbox\n# =====================\ndef resize_and_letterbox(img, new_size=640):\n    h, w = img.shape[:2]\n    scale = min(new_size / w, new_size / h)\n    new_w, new_h = int(w * scale), int(h * scale)\n\n    # Resize keeping aspect ratio\n    resized = cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_AREA)\n\n    # Create square canvas\n    canvas = np.full((new_size, new_size, 3), 0, dtype=np.uint8)\n    top = (new_size - new_h) // 2\n    left = (new_size - new_w) // 2\n    canvas[top:top+new_h, left:left+new_w] = resized\n\n    return canvas, scale, left, top\n\n# =====================\n# Load Data\n# =====================\nprint(\"Loading dataframes...\")\ndf_coords = pd.read_csv(input_base_path / 'train_label_coordinates.csv')\ndf_series_desc = pd.read_csv(input_base_path / 'train_series_descriptions.csv')\n\nprint(f\"Coords shape: {df_coords.shape}\")\nprint(f\"Series desc shape: {df_series_desc.shape}\")\n\ndf_coords_filtered = df_coords[df_coords['condition'].isin(class_map.keys())]\nprint(f\"After filtering: {df_coords_filtered.shape}\")\n\nif len(df_coords_filtered) > 0:\n    all_studies = df_coords_filtered['study_id'].unique()\n    print(f\"Unique studies: {len(all_studies)}\")\n\n    if len(all_studies) > 2:\n\n        train_studies, temp_studies = train_test_split(\n            all_studies, test_size=0.30, random_state=42\n        )\n        val_studies, test_studies = train_test_split(\n            temp_studies, test_size=0.50, random_state=42\n        )\n\n        train_df = df_coords_filtered[df_coords_filtered['study_id'].isin(train_studies)]\n        val_df   = df_coords_filtered[df_coords_filtered['study_id'].isin(val_studies)]\n        test_df  = df_coords_filtered[df_coords_filtered['study_id'].isin(test_studies)]\n\n        print(f\"Train labels: {len(train_df)} | Val: {len(val_df)} | Test: {len(test_df)}\")\n\n        # =====================\n        # Main Dataset Creator\n        # =====================\n        def create_yolo_dataset(dataframe, dataset_type='train'):\n            image_output_dir = output_base_path / dataset_type / 'images'\n            label_output_dir = output_base_path / dataset_type / 'labels'\n            image_output_dir.mkdir(parents=True, exist_ok=True)\n            label_output_dir.mkdir(parents=True, exist_ok=True)\n\n            processed_count = 0\n            grouped = dataframe.groupby(['study_id', 'series_id', 'instance_number'])\n\n            for (study_id, series_id, instance_number), group in grouped:\n                study_id_str = str(study_id)\n                series_id_str = str(series_id)\n                instance_number_str = str(instance_number)\n\n                dcm_path = input_base_path / 'train_images' / study_id_str / series_id_str / f\"{instance_number_str}.dcm\"\n                jpg_filename = f\"{study_id_str}_{series_id_str}_{instance_number_str}.jpg\"\n                txt_filename = f\"{study_id_str}_{series_id_str}_{instance_number_str}.txt\"\n\n                label_file_path = label_output_dir / txt_filename\n\n                if not dcm_path.exists():\n                    continue\n\n                try:\n                    ds = pydicom.dcmread(dcm_path)\n                    img_pixels = ds.pixel_array\n\n                    # DICOM windowing\n                    if hasattr(ds, 'WindowCenter') and hasattr(ds, 'WindowWidth'):\n                        center = ds.WindowCenter[0] if isinstance(ds.WindowCenter, pydicom.multival.MultiValue) else ds.WindowCenter\n                        width = ds.WindowWidth[0] if isinstance(ds.WindowWidth, pydicom.multival.MultiValue) else ds.WindowWidth\n                        low, high = center - width / 2, center + width / 2\n                        if high > low:\n                            img_pixels = np.clip(img_pixels, low, high)\n                            img_pixels = ((img_pixels - low) / (high - low) * 255).astype(np.uint8)\n                    else:\n                        img_pixels = ((img_pixels - img_pixels.min()) / (img_pixels.max() - img_pixels.min()) * 255).astype(np.uint8)\n\n                    # Convert grayscale → RGB\n                    if len(img_pixels.shape) == 2:\n                        img_pixels = cv2.cvtColor(img_pixels, cv2.COLOR_GRAY2RGB)\n\n                    # Resize to 640x640 with letterboxing\n                    resized_img, scale, left, top = resize_and_letterbox(img_pixels, new_size=TARGET_SIZE)\n\n                    # Save resized image\n                    jpg_path = image_output_dir / jpg_filename\n                    if not jpg_path.exists():\n                        cv2.imwrite(str(jpg_path), resized_img)\n\n                    img_height, img_width = TARGET_SIZE, TARGET_SIZE\n\n                except Exception as e:\n                    print(f\"Error {dcm_path}: {e}\")\n                    continue\n\n                # Collect YOLO labels\n                lines = []\n                for _, row in group.iterrows():\n                    if pd.isna(row['x']) or pd.isna(row['y']):\n                        continue\n\n                    # Original coords\n                    x_center, y_center = row['x'], row['y']\n\n                    # Apply scaling + padding\n                    x_center = x_center * scale + left\n                    y_center = y_center * scale + top\n\n                    # Fixed box size (scaled)\n                    w_pixels = BOX_WIDTH_RATIO * (img_width / scale) * scale\n                    h_pixels = BOX_HEIGHT_RATIO * (img_height / scale) * scale\n\n                    # Normalize to YOLO format\n                    x_norm = x_center / img_width\n                    y_norm = y_center / img_height\n                    w_norm = w_pixels / img_width\n                    h_norm = h_pixels / img_height\n\n                    class_id = class_map.get(row['condition'])\n                    if class_id is not None:\n                        lines.append(f\"{class_id} {x_norm:.6f} {y_norm:.6f} {w_norm:.6f} {h_norm:.6f}\\n\")\n\n                # Write label file\n                with open(label_file_path, 'w') as f:\n                    f.writelines(lines)\n\n                processed_count += 1\n                if processed_count % 1000 == 0:\n                    print(f\"Processed {processed_count} images for {dataset_type}\")\n\n            print(f\"Finished {dataset_type} set: {processed_count} images\")\n\n        # Run preprocessing\n        create_yolo_dataset(train_df, 'train')\n        create_yolo_dataset(val_df, 'val')\n        create_yolo_dataset(test_df, 'test')\n\n        # =====================\n        # Create dataset.yaml\n        # =====================\n        names_str = \"\\n\".join([f\"  - {n}\" for n in class_map.keys()])\n        yaml_content = f\"\"\"\npath: {output_base_path}\ntrain: train/images\nval: val/images\ntest: test/images\n\nnc: {len(class_map)}\n\nnames:\n{names_str}\n\"\"\"\n        yaml_file_path = output_base_path / \"dataset.yaml\"\n        with open(yaml_file_path, 'w') as f:\n            f.write(yaml_content.strip())\n\n        print(\"dataset.yaml created\")\n        print(yaml_content)\n\n    else:\n        print(\"Not enough studies for split\")\nelse:\n    print(\"No data after filtering\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T02:44:38.103273Z","iopub.execute_input":"2025-09-17T02:44:38.103509Z","iopub.status.idle":"2025-09-17T02:53:09.568999Z","shell.execute_reply.started":"2025-09-17T02:44:38.103491Z","shell.execute_reply":"2025-09-17T02:53:09.568239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required packages\n!pip install gdown\n!apt install -y git-lfs\n\n# First, create the zip file\nimport zipfile\nimport os\n\ndef zip_directory(path, zip_filename):\n    \"\"\"Zip a directory recursively\"\"\"\n    with zipfile.ZipFile(zip_filename, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, dirs, files in os.walk(path):\n            for file in files:\n                file_path = os.path.join(root, file)\n                arcname = os.path.relpath(file_path, os.path.dirname(path))\n                zipf.write(file_path, arcname)\n    return zip_filename\n\n# Create zip of your dataset\nzip_path = '/kaggle/working/yolo_dataset_update.zip'\nzip_directory('/kaggle/working/yolo_dataset', zip_path)\nprint(f\"Zip file created: {zip_path}\")\n\n# Upload to Google Drive using gdown (requires authentication)\nfrom google.colab import auth\nfrom googleapiclient.http import MediaFileUpload\nfrom googleapiclient.discovery import build\n\n# Authenticate and create drive service\nauth.authenticate_user()\ndrive_service = build('drive', 'v3')\n\n# Upload file\nfile_metadata = {'name': 'rsna_sciatica_detection.zip'}\nmedia = MediaFileUpload(zip_path, mimetype='application/zip')\nfile = drive_service.files().create(body=file_metadata, media_body=media, fields='id').execute()\n\nprint(f'File uploaded to Google Drive with ID: {file.get(\"id\")}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T02:55:29.344968Z","iopub.execute_input":"2025-09-17T02:55:29.345484Z","iopub.status.idle":"2025-09-17T03:04:27.424926Z","shell.execute_reply.started":"2025-09-17T02:55:29.345464Z","shell.execute_reply":"2025-09-17T03:04:27.423867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Upload to Google Drive using gdown (requires authentication)\nfrom google.colab import auth\nfrom googleapiclient.http import MediaFileUpload\nfrom googleapiclient.discovery import build\n\n# Authenticate and create drive service\nauth.authenticate_user()\ndrive_service = build('drive', 'v3')\n\n# Upload file\nfile_metadata = {'name': 'yolo_dataset_update.zip'}\nmedia = MediaFileUpload(zip_path, mimetype='application/zip')\nfile = drive_service.files().create(body=file_metadata, media_body=media, fields='id').execute()\n\nprint(f'File uploaded to Google Drive with ID: {file.get(\"id\")}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T03:11:23.353174Z","iopub.execute_input":"2025-09-17T03:11:23.353443Z","iopub.status.idle":"2025-09-17T03:12:04.917516Z","shell.execute_reply.started":"2025-09-17T03:11:23.353425Z","shell.execute_reply":"2025-09-17T03:12:04.916649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install ultralytics package\n!pip install ultralytics\nfrom ultralytics import YOLO\nimport torch\n\n# Path to your dataset.yaml\nyaml_file_path = '/kaggle/working/yolo_dataset/dataset.yaml'\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-09-10T08:30:01.874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install ultralytics package\n!pip install ultralytics\nfrom ultralytics import YOLO\nimport torch\n\n# Path to your dataset.yaml\nyaml_file_path = '/kaggle/working/yolo_dataset/dataset.yaml'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T08:26:58.215081Z","iopub.status.idle":"2025-09-10T08:26:58.215406Z","shell.execute_reply.started":"2025-09-10T08:26:58.215243Z","shell.execute_reply":"2025-09-10T08:26:58.215258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nimport torch\n\n# Set the path to your dataset YAML file\nyaml_file_path = '/kaggle/working/yolo_dataset/dataset.yaml'\n\n# Load model\nmodel = YOLO('yolov10s.pt')\n\n# Optimized training parameters\nresults = model.train(\n    data=yaml_file_path,\n    epochs=30,\n    imgsz=416,           # ⚡ Reduced from 640 (BIGGEST SAVINGS)\n    batch=8,             # ⚡ Reduced from 16\n    patience=15,\n    device=0,\n    workers=2,           # ⚡ Reduced from 4\n    project='rsna_sciatica_detection',\n    name='yolov10_fast',\n    optimizer='AdamW',   # Sometimes faster than auto\n    lr0=0.01,\n    lrf=0.01,\n    momentum=0.937,\n    weight_decay=0.0005,\n    warmup_epochs=2.0,   # ⚡ Reduced warmup\n    box=7.5,\n    cls=0.5,\n    dfl=1.5,\n    save=True,\n    save_period=25,      # ⚡ Save less frequently\n    val=False,           # ⚡⚡ NO validation during training (HUGE SAVINGS)\n    verbose=False,       # ⚡ Less console output\n)\n\n# Manual validation after training\nprint(\"Training complete! Running validation...\")\nmetrics = model.val()\nprint(f\"Final mAP50-95: {metrics.box.map:.4f}\")\nprint(f\"Final mAP50: {metrics.box.map50:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T08:22:26.821815Z","iopub.execute_input":"2025-09-10T08:22:26.822099Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run this in a new cell\nimport zipfile\nimport os\n\ndef zipdir(path, ziph):\n    # ziph is zipfile handle\n    for root, dirs, files in os.walk(path):\n        for file in files:\n            ziph.write(os.path.join(root, file), \n                       os.path.relpath(os.path.join(root, file), \n                                       os.path.join(path, '..')))\n\n# Create a zip file\nwith zipfile.ZipFile('/kaggle/working/val.zip', 'w', zipfile.ZIP_DEFLATED) as zipf:\n    zipdir('/kaggle/working/yolo_dataset/val', zipf)\n\nprint(\"Dataset zipped successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T18:27:21.116441Z","iopub.execute_input":"2025-09-09T18:27:21.117473Z","iopub.status.idle":"2025-09-09T18:27:33.294496Z","shell.execute_reply.started":"2025-09-09T18:27:21.117428Z","shell.execute_reply":"2025-09-09T18:27:33.293593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -la /kaggle/working/\n!ls -la /kaggle/working/yolo_dataset/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T14:55:18.516753Z","iopub.execute_input":"2025-09-09T14:55:18.517064Z","iopub.status.idle":"2025-09-09T14:55:18.799771Z","shell.execute_reply.started":"2025-09-09T14:55:18.51704Z","shell.execute_reply":"2025-09-09T14:55:18.798582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a direct download link\nfrom IPython.display import FileLink\nFileLink('/kaggle/working/yolo_dataset.zip')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T15:15:59.22286Z","iopub.execute_input":"2025-09-09T15:15:59.223284Z","iopub.status.idle":"2025-09-09T15:15:59.232608Z","shell.execute_reply.started":"2025-09-09T15:15:59.223249Z","shell.execute_reply":"2025-09-09T15:15:59.231589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required packages\n!pip install gdown\n!apt install -y git-lfs\n\n# First, create the zip file\nimport zipfile\nimport os\n\ndef zip_directory(path, zip_filename):\n    \"\"\"Zip a directory recursively\"\"\"\n    with zipfile.ZipFile(zip_filename, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, dirs, files in os.walk(path):\n            for file in files:\n                file_path = os.path.join(root, file)\n                arcname = os.path.relpath(file_path, os.path.dirname(path))\n                zipf.write(file_path, arcname)\n    return zip_filename\n\n# Create zip of your dataset\nzip_path = '/kaggle/working/yolo_dataset.zip'\nzip_directory('/kaggle/working/yolo_dataset', zip_path)\nprint(f\"Zip file created: {zip_path}\")\n\n# Upload to Google Drive using gdown (requires authentication)\nfrom google.colab import auth\nfrom googleapiclient.http import MediaFileUpload\nfrom googleapiclient.discovery import build\n\n# Authenticate and create drive service\nauth.authenticate_user()\ndrive_service = build('drive', 'v3')\n\n# Upload file\nfile_metadata = {'name': 'yolo_dataset.zip'}\nmedia = MediaFileUpload(zip_path, mimetype='application/zip')\nfile = drive_service.files().create(body=file_metadata, media_body=media, fields='id').execute()\n\nprint(f'File uploaded to Google Drive with ID: {file.get(\"id\")}')","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. First, create the zip file\nimport zipfile\nimport os\nfrom pathlib import Path\n\ndef zip_directory(path, zip_filename):\n    \"\"\"Zip a directory recursively\"\"\"\n    with zipfile.ZipFile(zip_filename, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, dirs, files in os.walk(path):\n            for file in files:\n                file_path = os.path.join(root, file)\n                # Add file to zip with relative path\n                arcname = os.path.relpath(file_path, os.path.dirname(path))\n                zipf.write(file_path, arcname)\n    return zip_filename\n\n# Create zip of your dataset\nzip_path = '/kaggle/working/yolo_dataset.zip'\nzip_directory('/kaggle/working/yolo_dataset', zip_path)\nprint(f\"Zip file created: {zip_path}\")\n\n# 2. Check file size\nfile_size = os.path.getsize(zip_path) / (1024 * 1024)  # Convert to MB\nprint(f\"Zip file size: {file_size:.2f} MB\")\n\n# 3. Create download link\nfrom IPython.display import FileLink, display\n\n# Method 1: Direct download link (works best in Kaggle)\nprint(\"\\n📥 Download your dataset:\")\ndisplay(FileLink(zip_path, result_html_prefix=\"Click here to download: \"))\n\n# Method 2: Alternative download approach\nprint(\"\\nAlternative download methods:\")\nprint(f\"1. Right-click this link and 'Save link as': [Download Dataset](/{zip_path})\")\nprint(\"2. Check the 'Data' tab on the right sidebar → /kaggle/working/ → download yolo_dataset.zip\")\n\n# 4. Verify zip file contents\nprint(\"\\n📁 Zip file contents preview:\")\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    file_list = zip_ref.namelist()\n    print(f\"Total files in zip: {len(file_list)}\")\n    print(\"First 10 files:\")\n    for file in file_list[:10]:\n        print(f\"  - {file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-09T18:19:29.592445Z","iopub.execute_input":"2025-09-09T18:19:29.59291Z","iopub.status.idle":"2025-09-09T18:20:29.683814Z","shell.execute_reply.started":"2025-09-09T18:19:29.592879Z","shell.execute_reply":"2025-09-09T18:20:29.682704Z"}},"outputs":[],"execution_count":null}]}