{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport pydicom\nfrom glob import glob\nimport matplotlib.patches as patches","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:16:03.092103Z","iopub.execute_input":"2025-08-05T06:16:03.092403Z","iopub.status.idle":"2025-08-05T06:16:07.947856Z","shell.execute_reply.started":"2025-08-05T06:16:03.092371Z","shell.execute_reply":"2025-08-05T06:16:07.946989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")\nclass_info = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv\")\n\ndf = pd.merge(labels, class_info, on=\"patientId\", how=\"left\")\ndf.sample(4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:18:01.941391Z","iopub.execute_input":"2025-08-05T06:18:01.941703Z","iopub.status.idle":"2025-08-05T06:18:02.050460Z","shell.execute_reply.started":"2025-08-05T06:18:01.941679Z","shell.execute_reply":"2025-08-05T06:18:02.049476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"NaN rows:\", df['class'].isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:20:03.902405Z","iopub.execute_input":"2025-08-05T06:20:03.902721Z","iopub.status.idle":"2025-08-05T06:20:03.910634Z","shell.execute_reply.started":"2025-08-05T06:20:03.902697Z","shell.execute_reply":"2025-08-05T06:20:03.909589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total Images:\", df['patientId'].nunique())\nprint(\"Total Rows:\", df.shape[0])\n# Plot target distribution\nsns.countplot(data=df, x='Target')\nplt.title('Pneumonia (1) vs No Pneumonia (0)')\nplt.show()\n# Unique patientId per class\nprint(\"Patients with Pneumonia:\", df[df['Target']==1]['patientId'].nunique())\nprint(\"Patients without Pneumonia:\", df[df['Target']==0]['patientId'].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:21:32.116295Z","iopub.execute_input":"2025-08-05T06:21:32.116581Z","iopub.status.idle":"2025-08-05T06:21:32.431269Z","shell.execute_reply.started":"2025-08-05T06:21:32.116562Z","shell.execute_reply":"2025-08-05T06:21:32.430264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=df, x='class')\nplt.title('Class Distribution')\nplt.xticks(rotation=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:23:12.829360Z","iopub.execute_input":"2025-08-05T06:23:12.829683Z","iopub.status.idle":"2025-08-05T06:23:13.010830Z","shell.execute_reply.started":"2025-08-05T06:23:12.829660Z","shell.execute_reply":"2025-08-05T06:23:13.009911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_dicom_image(patient_id, bbox=False):\n   dicom_path = f\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{patient_id}.dcm\"\n   ds = pydicom.dcmread(dicom_path)\n   fig, ax = plt.subplots(1, 1, figsize=(8, 8))\n   ax.imshow(ds.pixel_array, cmap='gray')\n   if bbox:\n       records = df[df['patientId'] == patient_id]\n       for _, row in records.iterrows():\n           if row['Target'] == 1:\n               rect = patches.Rectangle(\n                   (row['x'], row['y']), row['width'], row['height'],\n                   linewidth=2, edgecolor='red', facecolor='none'\n               )\n               ax.add_patch(rect)\n   plt.title(patient_id)\n   plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:24:55.964412Z","iopub.execute_input":"2025-08-05T06:24:55.964717Z","iopub.status.idle":"2025-08-05T06:24:55.971940Z","shell.execute_reply.started":"2025-08-05T06:24:55.964695Z","shell.execute_reply":"2025-08-05T06:24:55.970896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pneumonia_patients = df[df['Target']==1]['patientId'].unique()\nshow_dicom_image(pneumonia_patients[0], bbox=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T06:25:32.943833Z","iopub.execute_input":"2025-08-05T06:25:32.944179Z","iopub.status.idle":"2025-08-05T06:25:33.355571Z","shell.execute_reply.started":"2025-08-05T06:25:32.944119Z","shell.execute_reply":"2025-08-05T06:25:33.354599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q pydicom tensorflow-addons","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T12:44:28.041635Z","iopub.execute_input":"2025-08-05T12:44:28.041923Z","iopub.status.idle":"2025-08-05T12:44:33.749064Z","shell.execute_reply.started":"2025-08-05T12:44:28.041898Z","shell.execute_reply":"2025-08-05T12:44:33.748401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, pydicom, cv2, numpy as np, pandas as pd, tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import layers, models, callbacks, mixed_precision\n\n# Enable Mixed Precision\nmixed_precision.set_global_policy('mixed_float16')\n\nIMG_SIZE = 416\nBATCH_SIZE = 16\nEPOCHS = 20\nNUM_CLASSES = 1  # Only one class: Pneumonia\n\n# Load and prepare dataset\nlabel_df = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\nlabel_df = label_df[label_df['Target'] == 1]\nimg_dir = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'\ngrouped = label_df.groupby(\"patientId\").agg(list).reset_index()\n\n# Split dataset\ntrain_group, val_group = train_test_split(grouped, test_size=0.2, random_state=42)\n\n# Data Loader\ndef load_image_and_labels(row):\n    pid = row['patientId']\n    dicom_path = os.path.join(img_dir, pid + \".dcm\")\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array.astype(np.uint8)\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = np.stack([img]*3, axis=-1) / 255.0\n\n    w_ratio = IMG_SIZE / ds.Columns\n    h_ratio = IMG_SIZE / ds.Rows\n    boxes = []\n    for x, y, w, h in zip(row['x'], row['y'], row['width'], row['height']):\n        x1 = x * w_ratio\n        y1 = y * h_ratio\n        x2 = (x + w) * w_ratio\n        y2 = (y + h) * h_ratio\n        boxes.append([x1/IMG_SIZE, y1/IMG_SIZE, x2/IMG_SIZE, y2/IMG_SIZE])\n    if len(boxes) == 0:\n        boxes = [[0,0,0,0]]\n    return img, np.array(boxes[0], dtype=np.float32).reshape(-1, 4)\n\ndef make_dataset(group):\n    def generator():\n        for i in range(len(group)):\n            yield load_image_and_labels(group.iloc[i])\n    return tf.data.Dataset.from_generator(\n        generator,\n        output_signature=(\n            tf.TensorSpec(shape=(IMG_SIZE, IMG_SIZE, 3), dtype=tf.float32),\n            tf.TensorSpec(shape=(None, 4), dtype=tf.float32)\n        )\n    ).padded_batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\ntrain_dataset = make_dataset(train_group)\nval_dataset = make_dataset(val_group)\n\n# Model\ndef build_yolo(input_shape=(IMG_SIZE, IMG_SIZE, 3), num_classes=NUM_CLASSES):\n    inputs = tf.keras.Input(shape=input_shape)\n    x = layers.Conv2D(32, 3, strides=1, padding=\"same\", activation=\"relu\")(inputs)\n    x = layers.MaxPooling2D(2)(x)\n    x = layers.Conv2D(64, 3, padding=\"same\", activation=\"relu\")(x)\n    x = layers.MaxPooling2D(2)(x)\n    x = layers.Conv2D(128, 3, padding=\"same\", activation=\"relu\")(x)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dense(128, activation=\"relu\")(x)\n    x = layers.Dense(4, dtype=\"float32\")(x)  # [x1, y1, x2, y2]\n    return tf.keras.Model(inputs, x)\n\n# Compile\nmodel = build_yolo()\nmodel.compile(optimizer='adam', loss='mse', metrics=['mae'])\n\n# Callbacks\nckpt = callbacks.ModelCheckpoint(\"best_model.h5\", save_best_only=True, monitor='val_loss')\nearly = callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n\n# Train\nwith tf.device('/GPU:0'):\n    history = model.fit(\n        train_dataset,\n        validation_data=val_dataset,\n        epochs=EPOCHS,\n        callbacks=[ckpt, early]\n    )\n\n# Evaluation\nval_preds = []\nfor i, row in val_group.iterrows():\n    img, _ = load_image_and_labels(row)\n    pred = model.predict(img[np.newaxis, ...])[0]\n    x1, y1, x2, y2 = pred * IMG_SIZE\n    val_preds.append({\n        \"patientId\": row['patientId'],\n        \"x\": x1, \"y\": y1,\n        \"width\": x2 - x1,\n        \"height\": y2 - y1\n    })\n\n# Export\npd.DataFrame(val_preds).to_csv(\"rsna_val_predictions.csv\", index=False)\nprint(\"Exported results to rsna_val_predictions.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T13:07:56.916144Z","iopub.execute_input":"2025-08-05T13:07:56.916762Z","iopub.status.idle":"2025-08-05T13:28:20.265951Z","shell.execute_reply.started":"2025-08-05T13:07:56.916734Z","shell.execute_reply":"2025-08-05T13:28:20.265338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}