{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Step 1: Load both CSVs\ntrain_labels = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\nclass_info = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\n\n# Step 2: Merge on patientId\ndf = pd.merge(train_labels, class_info, on='patientId', how='left')\n\n# Step 3: Check the merged DataFrame\nprint(df.head())\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:01:33.567001Z","iopub.execute_input":"2025-04-28T14:01:33.567162Z","iopub.status.idle":"2025-04-28T14:01:35.437052Z","shell.execute_reply.started":"2025-04-28T14:01:33.567146Z","shell.execute_reply":"2025-04-28T14:01:35.436181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pydicom\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\n# Paths\nIMAGES_DIR = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images'\nOUTPUT_DIR = '/kaggle/working/pneumonia_yolo'\n\n# Create folders\nos.makedirs(f'{OUTPUT_DIR}/images/train', exist_ok=True)\nos.makedirs(f'{OUTPUT_DIR}/images/val', exist_ok=True)\nos.makedirs(f'{OUTPUT_DIR}/labels/train', exist_ok=True)\nos.makedirs(f'{OUTPUT_DIR}/labels/val', exist_ok=True)\n\n# Map classes\nclass_mapping = {\n    'Normal': 0,\n    'No Lung Opacity / Not Normal': 1,\n    'Lung Opacity': 2\n}\n\n# Select unique image IDs\nimage_ids = df['patientId'].unique()\n\n# [OPTIONAL]: Subsample if needed to save time (take first 6000 images)\nimage_ids = image_ids[:6000]\n\n# Split into train and validation\ntrain_ids, val_ids = train_test_split(image_ids, test_size=0.2, random_state=42)\n\n# Function to process each split\ndef process_images(ids, split='train'):\n    for img_id in tqdm(ids):\n        img_path = os.path.join(IMAGES_DIR, img_id + '.dcm')\n        \n        try:\n            # Load DICOM image\n            dicom = pydicom.dcmread(img_path)\n            img = dicom.pixel_array\n            img = Image.fromarray(img).convert('RGB')\n        except Exception as e:\n            print(f\"Error loading image {img_id}: {e}\")\n            continue\n        \n        # Resize image\n        img = img.resize((512, 512))\n        \n        # Save image\n        save_img_path = f'{OUTPUT_DIR}/images/{split}/{img_id}.jpg'\n        img.save(save_img_path)\n        \n        # Prepare label lines\n        records = df[df['patientId'] == img_id]\n        \n        label_lines = []\n        for idx, row in records.iterrows():\n            if row['class'] == 'Normal':\n                continue  # No boxes for Normal\n\n            x = row['x']\n            y = row['y']\n            w = row['width']\n            h = row['height']\n            \n            # Coordinates are in 1024x1024 space\n            x_center = (x + w/2) / 1024\n            y_center = (y + h/2) / 1024\n            w_norm = w / 1024\n            h_norm = h / 1024\n            \n            class_id = class_mapping[row['class']]\n            \n            label_lines.append(f\"{class_id} {x_center:.6f} {y_center:.6f} {w_norm:.6f} {h_norm:.6f}\")\n        \n        # Save labels\n        save_lbl_path = f'{OUTPUT_DIR}/labels/{split}/{img_id}.txt'\n        with open(save_lbl_path, 'w') as f:\n            for line in label_lines:\n                f.write(line + '\\n')\n\n# Process Train and Val splits\nprocess_images(train_ids, split='train')\nprocess_images(val_ids, split='val')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:06:31.159010Z","iopub.execute_input":"2025-04-28T14:06:31.159589Z","iopub.status.idle":"2025-04-28T14:09:17.401994Z","shell.execute_reply.started":"2025-04-28T14:06:31.159568Z","shell.execute_reply":"2025-04-28T14:09:17.401374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create YAML content\nyaml_content = \"\"\"\npath: /kaggle/working/pneumonia_yolo\ntrain: images/train\nval: images/val\n\nnc: 3\nnames: ['Normal', 'No Opacity', 'Opacity']\n\"\"\"\n\n# Save it\nwith open('/kaggle/working/pneumonia_yolo/pneumonia.yaml', 'w') as f:\n    f.write(yaml_content)\n\nprint(\"✅ pneumonia.yaml created successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:10:56.280204Z","iopub.execute_input":"2025-04-28T14:10:56.280592Z","iopub.status.idle":"2025-04-28T14:10:56.285931Z","shell.execute_reply.started":"2025-04-28T14:10:56.280564Z","shell.execute_reply":"2025-04-28T14:10:56.285164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q ultralytics\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:11:50.944772Z","iopub.execute_input":"2025-04-28T14:11:50.945521Z","iopub.status.idle":"2025-04-28T14:13:00.242625Z","shell.execute_reply.started":"2025-04-28T14:11:50.945488Z","shell.execute_reply":"2025-04-28T14:13:00.241690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load YOLO model\nmodel = YOLO('yolov8n.pt')  # nano model for faster training, or yolov8s.pt for small model\n\n# Start training\nmodel.train(\n    data='/kaggle/working/pneumonia_yolo/pneumonia.yaml',  # Our YAML file\n    imgsz=512,\n    epochs=30,\n    batch=16,\n    name='pneumonia_yolov8n',\n    workers=3,  # Kaggle gives 2 CPUs typically\n    patience=5,\n    optimizer='Adam',  # or 'Adam'\n    verbose=True\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:15:38.716078Z","iopub.execute_input":"2025-04-28T14:15:38.716418Z","iopub.status.idle":"2025-04-28T14:30:05.805215Z","shell.execute_reply.started":"2025-04-28T14:15:38.716368Z","shell.execute_reply":"2025-04-28T14:30:05.804430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming df is your merged CSV (from Part 1)\n# and train_ids, val_ids are your validation ids.\n\nsubset_df = df[df['patientId'].isin(val_ids.tolist())]\n\n# Create subsets\nnormal_ids = subset_df[subset_df['class'] == 'Normal']['patientId'].unique()\nno_opacity_ids = subset_df[subset_df['class'] == 'No Lung Opacity / Not Normal']['patientId'].unique()\nopacity_ids = subset_df[subset_df['class'] == 'Lung Opacity']['patientId'].unique()\n\nprint(f\"Normal images available: {len(normal_ids)}\")\nprint(f\"No Opacity images available: {len(no_opacity_ids)}\")\nprint(f\"Opacity images available: {len(opacity_ids)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:49:50.241904Z","iopub.execute_input":"2025-04-28T14:49:50.242560Z","iopub.status.idle":"2025-04-28T14:49:50.258545Z","shell.execute_reply.started":"2025-04-28T14:49:50.242536Z","shell.execute_reply":"2025-04-28T14:49:50.257765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport random\nimport os\nfrom ultralytics import YOLO\n\n# Load model\nmodel = YOLO('/kaggle/working/runs/detect/pneumonia_yolov8n/weights/best.pt')\n\n# Custom class names\ncustom_names = {0: \"Normal\", 1: \"No Opacity\", 2: \"Opacity\"}\n\n# Validation image directory\nval_dir = '/kaggle/working/pneumonia_yolo/images/val/'\n\n# Function to test a few images from a class\ndef test_class_images(patient_ids, label_name, sample_size=5):\n    sampled = random.sample(list(patient_ids), min(sample_size, len(patient_ids)))\n    \n    for pid in sampled:\n        img_path = os.path.join(val_dir, pid + '.jpg')\n        \n        results = model(img_path, conf=0.25)\n        preds = results[0].boxes\n\n        # Read image\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Draw boxes manually\n        for box, cls, conf in zip(preds.xyxy, preds.cls, preds.conf):\n            x1, y1, x2, y2 = map(int, box)\n            class_id = int(cls)\n            label = f\"{custom_names[class_id]} {conf:.2f}\"\n\n            # Draw red box\n            cv2.rectangle(img, (x1, y1), (x2, y2), (255, 0, 0), 2)\n            # Put label text\n            cv2.putText(img, label, (x1, y1-10), cv2.FONT_HERSHEY_SIMPLEX, \n                        0.5, (255, 0, 0), 2, cv2.LINE_AA)\n        \n        # Show results\n        plt.figure(figsize=(8,8))\n        plt.imshow(img)\n        plt.axis('off')\n        plt.title(f\"True Label: {label_name} | ID: {pid}\")\n        plt.show()\n\n# Example: Test Normal Images\nprint(\"Testing Normal Images:\")\ntest_class_images(normal_ids, label_name=\"Normal\")\n\n# Example: Test No Opacity Images\nprint(\"Testing No Opacity / Not Normal Images:\")\ntest_class_images(no_opacity_ids, label_name=\"No Opacity / Not Normal\")\n\n# Example: Test Opacity (Pneumonia) Images\nprint(\"Testing Opacity Images:\")\ntest_class_images(opacity_ids, label_name=\"Opacity\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T14:50:16.283031Z","iopub.execute_input":"2025-04-28T14:50:16.283701Z","iopub.status.idle":"2025-04-28T14:50:19.553893Z","shell.execute_reply.started":"2025-04-28T14:50:16.283664Z","shell.execute_reply":"2025-04-28T14:50:19.553202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\n\n# Create a download link for the best.pt model\nFileLink(r'/kaggle/working/runs/detect/pneumonia_yolov8n/weights/best.pt')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T15:09:10.205991Z","iopub.execute_input":"2025-04-28T15:09:10.206793Z","iopub.status.idle":"2025-04-28T15:09:10.211963Z","shell.execute_reply.started":"2025-04-28T15:09:10.206768Z","shell.execute_reply":"2025-04-28T15:09:10.211201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Load training results\nresults_csv = '/kaggle/working/runs/detect/pneumonia_yolov8n/results.csv'\ndf = pd.read_csv(results_csv)\n\n# Plot Losses\nplt.figure(figsize=(10,6))\nplt.plot(df['epoch'], df['train/box_loss'], label='Train Box Loss')\nplt.plot(df['epoch'], df['train/cls_loss'], label='Train Class Loss')\nplt.plot(df['epoch'], df['metrics/precision(B)'], label='Precision (B)')\nplt.plot(df['epoch'], df['metrics/recall(B)'], label='Recall (B)')\nplt.title('Losses, Precision and Recall over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Metric Value')\nplt.legend()\nplt.grid(True)\nplt.show()\n\n# Plot mAP (50) and mAP (50-95)\nplt.figure(figsize=(10,6))\nplt.plot(df['epoch'], df['metrics/mAP50(B)'], label='mAP@0.5')\nplt.plot(df['epoch'], df['metrics/mAP50-95(B)'], label='mAP@0.5:0.95')\nplt.title('Mean Average Precision over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('mAP Value')\nplt.legend()\nplt.grid(True)\nplt.show()\n\n# Plot Box Loss vs Epochs separately for clarity\nplt.figure(figsize=(8,5))\nplt.plot(df['epoch'], df['train/box_loss'], label='Box Loss', color='red')\nplt.title('Box Loss over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.show()\n\n# Plot Class Loss vs Epochs separately\nplt.figure(figsize=(8,5))\nplt.plot(df['epoch'], df['train/cls_loss'], label='Class Loss', color='blue')\nplt.title('Classification Loss over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T15:11:44.584955Z","iopub.execute_input":"2025-04-28T15:11:44.585585Z","iopub.status.idle":"2025-04-28T15:11:45.276610Z","shell.execute_reply.started":"2025-04-28T15:11:44.585561Z","shell.execute_reply":"2025-04-28T15:11:45.275826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Precision and Recall over Epochs\nplt.figure(figsize=(10,6))\nplt.plot(df['epoch'], df['metrics/precision(B)'], label='Precision (B)')\nplt.plot(df['epoch'], df['metrics/recall(B)'], label='Recall (B)')\nplt.title('Precision and Recall over Epochs (Threshold Stability Analysis)')\nplt.xlabel('Epoch')\nplt.ylabel('Metric Value')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T15:12:38.500148Z","iopub.execute_input":"2025-04-28T15:12:38.500699Z","iopub.status.idle":"2025-04-28T15:12:38.691844Z","shell.execute_reply.started":"2025-04-28T15:12:38.500675Z","shell.execute_reply":"2025-04-28T15:12:38.691058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nplt.plot(df['epoch'], df['metrics/mAP50(B)'], label='mAP@0.5')\nplt.title('Threshold Stability Analysis (mAP@0.5 vs Epoch)')\nplt.xlabel('Epoch')\nplt.ylabel('mAP@0.5')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T15:13:59.658534Z","iopub.execute_input":"2025-04-28T15:13:59.659292Z","iopub.status.idle":"2025-04-28T15:13:59.833767Z","shell.execute_reply.started":"2025-04-28T15:13:59.659266Z","shell.execute_reply":"2025-04-28T15:13:59.832972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}