{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q -U efficientnet tensorflow_addons==0.20.0 typeguard==2.13.3\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\nimport efficientnet.tfkeras as efn\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"editable":false,"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:24:07.072447Z","iopub.execute_input":"2025-07-17T11:24:07.072875Z","iopub.status.idle":"2025-07-17T11:24:10.401077Z","shell.execute_reply.started":"2025-07-17T11:24:07.072842Z","shell.execute_reply":"2025-07-17T11:24:10.399977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/rsna-pneumonia-detection-challenge/'\nlabels_df = pd.read_csv(os.path.join(data_dir, 'stage_2_train_labels.csv'))\nlabels_df = labels_df.drop_duplicates('patientId')\nlabels_df['Target'] = labels_df['Target'].astype(str)","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T11:24:13.926410Z","iopub.execute_input":"2025-07-17T11:24:13.926744Z","iopub.status.idle":"2025-07-17T11:24:14.016080Z","shell.execute_reply.started":"2025-07-17T11:24:13.926714Z","shell.execute_reply":"2025-07-17T11:24:14.015499Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_dir = os.path.join(data_dir, 'stage_2_train_images')\nlabels_df['path'] = labels_df['patientId'].apply(lambda x: os.path.join(image_dir, f\"{x}.dcm\"))","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T11:24:16.861375Z","iopub.execute_input":"2025-07-17T11:24:16.861700Z","iopub.status.idle":"2025-07-17T11:24:16.892944Z","shell.execute_reply.started":"2025-07-17T11:24:16.861673Z","shell.execute_reply":"2025-07-17T11:24:16.892346Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\ndef load_dicom_image(path, resize=(224, 224)):\n    dcm = pydicom.dcmread(path)\n    image = dcm.pixel_array\n    image = cv2.resize(image, resize)\n    image = np.stack((image,) * 3, axis=-1)  # Convert to 3 channels\n    image = image / 255.0\n    return image","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T11:24:18.941417Z","iopub.execute_input":"2025-07-17T11:24:18.941735Z","iopub.status.idle":"2025-07-17T11:24:19.373096Z","shell.execute_reply.started":"2025-07-17T11:24:18.941712Z","shell.execute_reply":"2025-07-17T11:24:19.372493Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PneumoniaDataset(tf.keras.utils.Sequence):\n    def __init__(self, df, batch_size=16, shuffle=True, augment=False, **kwargs):\n        super().__init__(**kwargs)\n        self.df = df\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.on_epoch_end()\n\n        \n    def __len__(self):\n        return int(np.floor(len(self.df) / self.batch_size))\n    \n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.df))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        df_batch = self.df.iloc[indexes]\n        \n        X = np.array([load_dicom_image(path) for path in df_batch['path']])\n        y = df_batch['Target'].astype(int).values\n        \n        if self.augment:\n            for i in range(len(X)):\n                if np.random.rand() < 0.5:\n                    X[i] = tf.image.flip_left_right(X[i])\n        \n        return X, y","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T11:24:21.501358Z","iopub.execute_input":"2025-07-17T11:24:21.501637Z","iopub.status.idle":"2025-07-17T11:24:21.508481Z","shell.execute_reply.started":"2025-07-17T11:24:21.501616Z","shell.execute_reply":"2025-07-17T11:24:21.507799Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(labels_df, test_size=0.2, stratify=labels_df['Target'], random_state=42)\ntrain_gen = PneumoniaDataset(train_df, batch_size=16, augment=True)\nval_gen = PneumoniaDataset(val_df, batch_size=16, augment=False)","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T11:24:24.821348Z","iopub.execute_input":"2025-07-17T11:24:24.821842Z","iopub.status.idle":"2025-07-17T11:24:24.859120Z","shell.execute_reply.started":"2025-07-17T11:24:24.821816Z","shell.execute_reply":"2025-07-17T11:24:24.858567Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetV2M\n\ndef build_model():\n    base_model = EfficientNetV2M(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n    base_model.trainable = False\n\n    inputs = keras.Input(shape=(224, 224, 3))\n    x = base_model(inputs, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.3)(x)\n    outputs = layers.Dense(1, activation='sigmoid')(x)\n\n    model = keras.Model(inputs, outputs)\n    model.compile(optimizer='adam',\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\nmodel = build_model()\nmodel.summary()","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T11:24:27.461536Z","iopub.execute_input":"2025-07-17T11:24:27.462398Z","iopub.status.idle":"2025-07-17T11:24:35.082339Z","shell.execute_reply.started":"2025-07-17T11:24:27.462361Z","shell.execute_reply":"2025-07-17T11:24:35.081697Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(train_gen,\n                    validation_data=val_gen,\n                    epochs=10)","metadata":{"editable":false,"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T11:27:26.511388Z","iopub.execute_input":"2025-07-17T11:27:26.511940Z","iopub.status.idle":"2025-07-17T12:07:32.956418Z","shell.execute_reply.started":"2025-07-17T11:27:26.511920Z","shell.execute_reply":"2025-07-17T12:07:32.955803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save model\nmodel.save(\"pneumonia_model.h5\")","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T12:08:47.342891Z","iopub.execute_input":"2025-07-17T12:08:47.343213Z","iopub.status.idle":"2025-07-17T12:08:48.640980Z","shell.execute_reply.started":"2025-07-17T12:08:47.343191Z","shell.execute_reply":"2025-07-17T12:08:48.640423Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel = load_model(\"pneumonia_model.h5\")","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T12:08:51.912372Z","iopub.execute_input":"2025-07-17T12:08:51.912640Z","iopub.status.idle":"2025-07-17T12:08:55.792353Z","shell.execute_reply.started":"2025-07-17T12:08:51.912618Z","shell.execute_reply":"2025-07-17T12:08:55.791693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig, axs = plt.subplots(1, 2, figsize=(14, 5))\n\n# Accuracy plot\naxs[0].plot(history.history['accuracy'], label='Train Accuracy')\naxs[0].plot(history.history['val_accuracy'], label='Val Accuracy')\naxs[0].set_title('Accuracy over Epochs')\naxs[0].set_xlabel('Epochs')\naxs[0].set_ylabel('Accuracy')\naxs[0].legend()\n\n# Loss plot\naxs[1].plot(history.history['loss'], label='Train Loss')\naxs[1].plot(history.history['val_loss'], label='Val Loss')\naxs[1].set_title('Loss over Epochs')\naxs[1].set_xlabel('Epochs')\naxs[1].set_ylabel('Loss')\naxs[1].legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T12:08:59.613345Z","iopub.execute_input":"2025-07-17T12:08:59.613590Z","iopub.status.idle":"2025-07-17T12:09:00.003795Z","shell.execute_reply.started":"2025-07-17T12:08:59.613573Z","shell.execute_reply":"2025-07-17T12:09:00.003102Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class Distribution\nlabels_df['Target'].value_counts().plot(kind='bar', color=['green', 'red'])\nplt.title(\"Data Distribution (0 = Normal, 1 = Pneumonia)\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Count\")\nplt.show()\nimport os\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport pydicom\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import load_model\n\n# Load labels and paths\ndata_dir = '/kaggle/input/rsna-pneumonia-detection-challenge/'\nlabels_df = pd.read_csv(os.path.join(data_dir, 'stage_2_train_labels.csv'))\nlabels_df = labels_df.drop_duplicates('patientId')\nlabels_df['Target'] = labels_df['Target'].astype(int)\nimage_dir = os.path.join(data_dir, 'stage_2_train_images')\nlabels_df['path'] = labels_df['patientId'].apply(lambda x: os.path.join(image_dir, f\"{x}.dcm\"))\n\n# Function to load and preprocess DICOM images\ndef load_dicom_image(path, resize=(224, 224)):\n    try:\n        dcm = pydicom.dcmread(path)\n        img = dcm.pixel_array\n        img = cv2.resize(img, resize)\n        img = np.stack((img,) * 3, axis=-1)\n        img = img / 255.0\n        return img\n    except:\n        return None\n\n# Load a small sample to avoid memory overload\nsample_df = labels_df.sample(n=1000, random_state=42)\nsample_df['image'] = sample_df['path'].apply(load_dicom_image)\nsample_df = sample_df.dropna(subset=['image'])\n\n# Prepare X and y\nX = np.stack(sample_df['image'].values)\ny = sample_df['Target'].values\n\n# Split and load model\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\nmodel = load_model(\"pneumonia_model.h5\")\n\n# Predict and evaluate\ny_pred = model.predict(X_test)\ny_pred_labels = (y_pred > 0.5).astype(int)\n\n# Accuracy\nacc = accuracy_score(y_test, y_pred_labels)\nprint(\"✅ Overall Accuracy on Test Set:\", acc)","metadata":{"editable":false,"execution":{"iopub.status.busy":"2025-07-17T12:09:17.254091Z","iopub.execute_input":"2025-07-17T12:09:17.254371Z","iopub.status.idle":"2025-07-17T12:10:00.885419Z","shell.execute_reply.started":"2025-07-17T12:09:17.254350Z","shell.execute_reply":"2025-07-17T12:10:00.884828Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import (\n    accuracy_score, f1_score, precision_score, recall_score,\n    confusion_matrix, classification_report, ConfusionMatrixDisplay\n)\nimport matplotlib.pyplot as plt\n\n# Predict\ny_pred = model.predict(X_test)\ny_pred_labels = (y_pred > 0.5).astype(int)\n\n# Metrics\nacc = accuracy_score(y_test, y_pred_labels)\nf1 = f1_score(y_test, y_pred_labels)\nprecision = precision_score(y_test, y_pred_labels)\nrecall = recall_score(y_test, y_pred_labels)\n\n# Output metrics\nprint(f\"✅ Accuracy:  {acc:.4f}\")\n\n\n# Classification Report\nprint(\"\\n📋 Classification Report:\\n\")\nprint(classification_report(y_test, y_pred_labels))\n\n# Confusion Matrix\ncm = confusion_matrix(y_test, y_pred_labels)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=[\"Normal\", \"Pneumonia\"])\ndisp.plot(cmap=plt.cm.Blues)\nplt.title(\"🧠 Confusion Matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:10:08.517573Z","iopub.execute_input":"2025-07-17T12:10:08.517882Z","iopub.status.idle":"2025-07-17T12:10:09.754387Z","shell.execute_reply.started":"2025-07-17T12:10:08.517859Z","shell.execute_reply":"2025-07-17T12:10:09.753506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import load_model\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport os\nimport pydicom\n\n# ====== Paths ======\nmodel_path = '/kaggle/working/pneumonia_model.h5'\nimage_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images/0005d3cc-3c3f-40b9-93c3-46231c3eb813.dcm'\n\n# ====== Load Model ======\nmodel = load_model(model_path)\n\n# ====== Load and Preprocess ======\ndef load_image(img_path, target_size=(224, 224)):\n    ext = os.path.splitext(img_path)[1].lower()\n\n    if ext == '.dcm':\n        dicom = pydicom.dcmread(img_path)\n        img = dicom.pixel_array\n        img = cv2.normalize(img, None, 0, 255, cv2.NORM_MINMAX)\n        img = cv2.cvtColor(np.uint8(img), cv2.COLOR_GRAY2RGB)\n    else:\n        img = Image.open(img_path).convert('RGB')\n        img = np.array(img)\n\n    img = cv2.resize(img, target_size)\n    img = img / 255.0\n    return img\n\n# ====== Predict and Show ======\ndef predict_and_show(img_path):\n    img_array = load_image(img_path)\n    input_array = np.expand_dims(img_array, axis=0)\n    pred = model.predict(input_array)[0][0]\n\n    label = \"PNEUMONIA\" if pred > 0.5 else \"NORMAL\"\n\n    display_img = (img_array * 255).astype(np.uint8)\n    green = (0, 255, 0)\n    display_img = cv2.putText(display_img.copy(), label, (10, 30),\n                              cv2.FONT_HERSHEY_SIMPLEX, 1, green, 2)\n\n    plt.imshow(display_img)\n    plt.axis('off')\n    plt.show()\n\n# 🔍 Run prediction\npredict_and_show(image_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:14:04.424086Z","iopub.execute_input":"2025-07-17T12:14:04.424989Z","iopub.status.idle":"2025-07-17T12:14:21.390520Z","shell.execute_reply.started":"2025-07-17T12:14:04.424961Z","shell.execute_reply":"2025-07-17T12:14:21.389788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nfrom IPython.display import FileLink\n\n# ====== Zip the model ======\nshutil.make_archive('pneumonia_model', 'zip', '/kaggle/working', 'pneumonia_model.h5')\n\n# ====== Create Download Link ======\nFileLink('pneumonia_model.zip')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-17T12:16:22.049288Z","iopub.execute_input":"2025-07-17T12:16:22.050203Z","iopub.status.idle":"2025-07-17T12:16:33.934849Z","shell.execute_reply.started":"2025-07-17T12:16:22.050158Z","shell.execute_reply":"2025-07-17T12:16:33.934094Z"}},"outputs":[],"execution_count":null}]}