{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":104953,"sourceType":"datasetVersion","datasetId":54936},{"sourceId":4992989,"sourceType":"datasetVersion","datasetId":2552962}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  1- Imports & Dataset Paths\n\nIn this cell, we import all the required libraries for data handling, visualization, and image processing.  \nWe also define the dataset paths for RSNA, NIH, and NLP datasets, along with the RSNA labels CSV file.","metadata":{}},{"cell_type":"code","source":"!pip install -q pydicom wordcloud\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:13:44.816914Z","iopub.execute_input":"2025-10-08T12:13:44.817185Z","iopub.status.idle":"2025-10-08T12:13:47.935837Z","shell.execute_reply.started":"2025-10-08T12:13:44.817162Z","shell.execute_reply":"2025-10-08T12:13:47.935018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n\nimport time\n\n# Optional: redirect stderr\nimport sys\nsys.stderr = open(os.devnull, 'w')\n\nimport tensorflow as tf\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib.patches as patches\nimport warnings\n\nfrom PIL import Image\nfrom pathlib import Path\nfrom wordcloud import WordCloud\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, Flatten,MaxPooling2D,Conv2D\nfrom tensorflow.keras.models import Model,Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\n\nwarnings.filterwarnings('ignore')\n%matplotlib inline\n\n\nBASE_DIR = Path('/kaggle/input/rsna-pneumonia-detection-challenge')\nDICOM_DIR = BASE_DIR / 'stage_2_train_images'\nPNG_DIR = Path('/kaggle/input/rsna-pneu-train-png/orig')\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-10-08T12:13:56.013404Z","iopub.execute_input":"2025-10-08T12:13:56.014180Z","iopub.status.idle":"2025-10-08T12:13:56.022207Z","shell.execute_reply.started":"2025-10-08T12:13:56.014155Z","shell.execute_reply":"2025-10-08T12:13:56.021638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Load and Prepare Data\n","metadata":{}},{"cell_type":"markdown","source":"1.  **Check Directories**: First, we verify that both the original RSNA competition dataset (containing labels) and the PNG image dataset are attached to the notebook.\n2.  **Load Labels**: We load the `stage_2_train_labels.csv` file.\n3.  **Create Classification DataFrame**: We create a clean `df_class` DataFrame with one unique row per patient, linking each `patientId` to its corresponding `.png` filename.\n","metadata":{}},{"cell_type":"code","source":"print(\"=== Preparing RSNA Classification Data ===\")\n\n# --- 1. Check for required directories ---\nif not BASE_DIR.exists() :\n    print(f\"❌ Error: A required directory was not found.\")\n    print(\"Please ensure 'rsna-pneumonia-detection-challenge' datasets is added.\")\nelse:\n    print(f\"\\n✅ The required dataset directory is found.\")\n    \n    # --- 2. Load Label Data ---\n    labels_path = BASE_DIR / 'stage_2_train_labels.csv'\n    if labels_path.exists():\n        df_labels = pd.read_csv(labels_path)\n        print(f\"\\n✅ Success! RSNA Labels loaded. Shape: {df_labels.shape}\")\n\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:13:59.320313Z","iopub.execute_input":"2025-10-08T12:13:59.320701Z","iopub.status.idle":"2025-10-08T12:13:59.382029Z","shell.execute_reply.started":"2025-10-08T12:13:59.320674Z","shell.execute_reply":"2025-10-08T12:13:59.381440Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"4. **Understand the Dataset Of The Labels With Basic EDA** ","metadata":{}},{"cell_type":"code","source":"df_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:14:09.906303Z","iopub.execute_input":"2025-10-08T12:14:09.906806Z","iopub.status.idle":"2025-10-08T12:14:09.930325Z","shell.execute_reply.started":"2025-10-08T12:14:09.906778Z","shell.execute_reply":"2025-10-08T12:14:09.929888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n🔍 Null Value Check\")\nprint(\"-\" * 60)\nprint(df_labels.isnull().sum())\n\nprint(\"\\n📎 Duplicate Check\")\nprint(\"-\" * 60)\ndup_count = df_labels.duplicated().sum()\nprint(df_labels.duplicated('patientId').sum())\nprint(f\"Total duplicate rows: {dup_count}\")\n\nprint(\"\\nℹ️ Basic Data Information\")\nprint(\"-\" * 60)\ndf_labels.info()\n\nprint(\"\\n📊 Summary Statistics (Numerical Columns)\")\nprint(\"-\" * 60)\ndisplay(df_labels.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:14:12.437366Z","iopub.execute_input":"2025-10-08T12:14:12.438058Z","iopub.status.idle":"2025-10-08T12:14:12.495492Z","shell.execute_reply.started":"2025-10-08T12:14:12.438030Z","shell.execute_reply":"2025-10-08T12:14:12.494961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df_labels['Target'].value_counts().sort_index()\nprint(df_labels['Target'].value_counts())\n\nclass_counts.plot(kind='bar', color=['teal', 'indigo'])\nplt.title(\"Pneumonia vs Normal Cases\")\nplt.xlabel(\"Target (0=Normal, 1=Pneumonia)\")\nplt.ylabel(\"Count\")\nplt.xticks(rotation=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:14:16.108941Z","iopub.execute_input":"2025-10-08T12:14:16.109639Z","iopub.status.idle":"2025-10-08T12:14:16.384387Z","shell.execute_reply.started":"2025-10-08T12:14:16.109614Z","shell.execute_reply":"2025-10-08T12:14:16.383752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"5.  **Verify Files**: We cross-reference this DataFrame with the actual files in the `PNG_DIR` to ensure we only work with images that exist, preventing \"File Not Found\" errors during training.","metadata":{}},{"cell_type":"code","source":"# --- 3. Create and Verify Classification Data ---\n# This creates a dataframe with one row per unique patient for classification\ndf_class = df_labels.drop_duplicates('patientId')[['patientId', 'Target']].copy()\ndf_class['filename'] = df_class['patientId'].apply(lambda x: f\"{x}.png\")\ndf_class['Target'] = df_class['Target'].astype(str)  # Convert target to string for the generator\n\n# Verify that the corresponding PNG file exists for each entry\nprint(\"\\nVerifying image files exist...\")\ndf_class['filepath'] = df_class['filename'].apply(lambda f: PNG_DIR / f)\ndf_class['file_exists'] = df_class['filepath'].apply(lambda p: p.exists())\n\n# Filter out records where the image file is missing\ndf_class = df_class[df_class['file_exists']].copy().reset_index(drop=True)\n\nprint(f\"\\nFound {len(df_class)} images with corresponding labels.\")\nprint(\"Dataframe ready for EDA and preprocessing:\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:14:22.069150Z","iopub.execute_input":"2025-10-08T12:14:22.069437Z","iopub.status.idle":"2025-10-08T12:15:26.482294Z","shell.execute_reply.started":"2025-10-08T12:14:22.069416Z","shell.execute_reply":"2025-10-08T12:15:26.481603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_class.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:16:34.944710Z","iopub.execute_input":"2025-10-08T12:16:34.945313Z","iopub.status.idle":"2025-10-08T12:16:34.954608Z","shell.execute_reply.started":"2025-10-08T12:16:34.945286Z","shell.execute_reply":"2025-10-08T12:16:34.954082Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Exploratory Data Analysis (EDA)\n\nWith the data successfully loaded, we now explore its characteristics.\n\n\n","metadata":{}},{"cell_type":"markdown","source":"1.  **Class Distribution**: We visualize the count of \"Normal\" vs. \"Pneumonia\" cases to understand the class imbalance in the dataset.","metadata":{}},{"cell_type":"code","source":"# --- 1. Visualize the Class Distribution ---\nplt.figure(figsize=(8, 6))\nsns.countplot(x='Target', data=df_class, palette='viridis')\nplt.title('Distribution of Pneumonia Cases (0 = Normal, 1 = Pneumonia)', fontsize=14)\nplt.xlabel('Class', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.show()\n\nprint(\"Class Counts:\")\nprint(df_class['Target'].value_counts())\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:16:39.909651Z","iopub.execute_input":"2025-10-08T12:16:39.909961Z","iopub.status.idle":"2025-10-08T12:16:40.052772Z","shell.execute_reply.started":"2025-10-08T12:16:39.909939Z","shell.execute_reply":"2025-10-08T12:16:40.052205Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"2.  **Metadata Extraction**: We extract valuable metadata (Age, Sex, View Position) from the original DICOM files. This step is crucial for conducting a fairness analysis later to ensure the model performs equitably across different demographic groups.","metadata":{}},{"cell_type":"code","source":"# --- 2. Add Metadata for Fairness Analysis ---\nprint(\"\\nExtracting metadata from DICOM files for fairness analysis...\")\n# This will take a few minutes to run\nages, sexes, view_positions = [], [], []\n\nstart_time = time.time()  # Start timer\n\nfor patient_id in tqdm(df_class['patientId']):\n    dcm_path = DICOM_DIR / f\"{patient_id}.dcm\"\n    dcm_data = pydicom.dcmread(dcm_path, stop_before_pixels=True)\n    ages.append(dcm_data.PatientAge)\n    sexes.append(dcm_data.PatientSex)\n    view_positions.append(dcm_data.ViewPosition)\n\ndf_class['Age'] = ages\ndf_class['Sex'] = sexes\ndf_class['ViewPosition'] = view_positions\ndf_class['Age'] = df_class['Age'].astype(int)\ndf_class['AgeGroup'] = pd.cut(df_class['Age'], bins=[0, 18, 40, 60, 150], labels=['Child', 'Young Adult', 'Adult', 'Senior'])\n\nend_time = time.time()  # End timer\nelapsed = end_time - start_time\nprint(f\"\\n✅ Metadata extraction complete in {elapsed:.2f} seconds.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:16:58.770672Z","iopub.execute_input":"2025-10-08T12:16:58.770970Z","iopub.status.idle":"2025-10-08T12:19:26.855187Z","shell.execute_reply.started":"2025-10-08T12:16:58.770948Z","shell.execute_reply":"2025-10-08T12:19:26.854585Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3.  **Metadata Distribution**: We visualize the distribution of patients by sex and age group.","metadata":{}},{"cell_type":"code","source":"# --- 3. Analyze Metadata Distribution ---\nfig, axes = plt.subplots(1, 2, figsize=(16, 6))\nsns.countplot(x='Sex', data=df_class, ax=axes[0], palette='magma')\naxes[0].set_title('Distribution by Sex')\nsns.countplot(x='AgeGroup', data=df_class, ax=axes[1], palette='plasma')\naxes[1].set_title('Distribution by Age Group')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:19:39.897380Z","iopub.execute_input":"2025-10-08T12:19:39.898107Z","iopub.status.idle":"2025-10-08T12:19:40.150278Z","shell.execute_reply.started":"2025-10-08T12:19:39.898086Z","shell.execute_reply":"2025-10-08T12:19:40.149769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Visualizing Image Examples\n\nTo get a better understanding of the data, let's look at some example chest X-ray images from our dataset. We will display a few samples for both **Normal** (Target = 0) and **Pneumonia** (Target = 1) cases. This helps in visually confirming the characteristics the model will learn to distinguish.","metadata":{}},{"cell_type":"code","source":"\n\n# --- Function to plot image samples ---\ndef plot_image_samples(df, target, num_samples=5, figsize=(20, 5)):\n    sample_df = df[df['Target'] == str(target)].sample(num_samples, random_state=42)\n    \n    fig, axes = plt.subplots(1, num_samples, figsize=figsize)\n    fig.suptitle(f'Sample Images: {\"Pneumonia\" if target == 1 else \"Normal\"} Cases', fontsize=16)\n    \n    for i, (idx, row) in enumerate(sample_df.iterrows()):\n        img_path = row['filepath']\n        img = Image.open(img_path).convert('RGB') # Convert to RGB for consistency\n        axes[i].imshow(img, cmap='gray')\n        axes[i].set_title(f\"Patient: {row['patientId'][:8]}...\", fontsize=10)\n        axes[i].axis('off')\n\n        # If it's a pneumonia case, draw the bounding box(es)\n        if target == 1:\n            bboxes = df_labels[df_labels['patientId'] == row['patientId']]\n            for _, box_row in bboxes.iterrows():\n                if not np.isnan(box_row['x']):\n                    # Create a Rectangle patch\n                    rect = patches.Rectangle(\n                        (box_row['x'], box_row['y']),\n                        box_row['width'],\n                        box_row['height'],\n                        linewidth=2,\n                        edgecolor='r',\n                        facecolor='none'\n                    )\n                    # Add the patch to the Axes\n                    axes[i].add_patch(rect)\n    plt.tight_layout(rect=[0, 0, 1, 0.95])\n    plt.show()\n\n# --- Display the images ---\nprint(\"Displaying sample images...\")\n\n# Display Normal cases (Target = 0)\nplot_image_samples(df_class, target=0, num_samples=5)\n\n# Display Pneumonia cases (Target = 1)\nplot_image_samples(df_class, target=1, num_samples=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:20:08.914924Z","iopub.execute_input":"2025-10-08T12:20:08.915619Z","iopub.status.idle":"2025-10-08T12:20:11.352862Z","shell.execute_reply.started":"2025-10-08T12:20:08.915596Z","shell.execute_reply":"2025-10-08T12:20:11.352203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Data Preprocessing & Splitting\n\nThis is the final preparation step before we build our model. We split the master DataFrame (`df_class`) into training and validation sets.\n\n- **Stratification**: We use the `stratify` option to ensure that the proportion of normal-to-pneumonia cases is the same in both the training (`df_train`) and validation (`df_val`) sets. This is a critical best practice for imbalanced datasets as it leads to more reliable model evaluation.","metadata":{}},{"cell_type":"code","source":"print(\"\\n=== Data Preprocessing & Splitting ===\")\n\ndf_train, df_val = train_test_split(\n    df_class,\n    test_size=0.2, # Using 20% of the data for validation\n    random_state=42,\n    stratify=df_class['Target'] # Stratify ensures similar class distribution in both sets\n)\n\nprint(f\"\\nData successfully split:\")\nprint(f\"Training set size:   {len(df_train)}\")\nprint(f\"Validation set size: {len(df_val)}\")\nprint(\"\\nTraining Set Class Distribution:\")\nprint(df_train['Target'].value_counts(normalize=True))\nprint(\"\\nValidation Set Class Distribution:\")\nprint(df_val['Target'].value_counts(normalize=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:21:56.183103Z","iopub.execute_input":"2025-10-08T12:21:56.183483Z","iopub.status.idle":"2025-10-08T12:21:56.224694Z","shell.execute_reply.started":"2025-10-08T12:21:56.183461Z","shell.execute_reply":"2025-10-08T12:21:56.224177Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Model Preparation: Data Augmentation & Generators\n\nBefore we feed our images to the model, we need to process them. We will use Keras's `ImageDataGenerator` for this.","metadata":{}},{"cell_type":"markdown","source":"###  Step 1: Define Image Parameters\nWe first set the image size and batch size — these control the shape and number of images fed to the model at once.\n","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 32","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:25:12.879062Z","iopub.execute_input":"2025-10-08T12:25:12.879770Z","iopub.status.idle":"2025-10-08T12:25:12.882389Z","shell.execute_reply.started":"2025-10-08T12:25:12.879744Z","shell.execute_reply":"2025-10-08T12:25:12.881900Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 2: Create Data Augmentation Generator for Training\nThe training generator will apply random transformations (rotation, zoom, flipping, etc.)\nto make the model more robust and reduce overfitting.\n","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1./255.,\n    rotation_range=15,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:25:16.142092Z","iopub.execute_input":"2025-10-08T12:25:16.142634Z","iopub.status.idle":"2025-10-08T12:25:16.145896Z","shell.execute_reply.started":"2025-10-08T12:25:16.142609Z","shell.execute_reply":"2025-10-08T12:25:16.145277Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 3: Create a Simple Rescaling Generator for Validation\nValidation data should **not** be augmented — we only rescale pixel values to `[0, 1]`\nto evaluate the model on clean, unmodified images.\n","metadata":{}},{"cell_type":"code","source":"val_datagen = ImageDataGenerator(rescale=1./255.)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:25:22.747456Z","iopub.execute_input":"2025-10-08T12:25:22.748169Z","iopub.status.idle":"2025-10-08T12:25:22.751148Z","shell.execute_reply.started":"2025-10-08T12:25:22.748145Z","shell.execute_reply":"2025-10-08T12:25:22.750610Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 4: Create Generators That Flow Data from the DataFrame\nNow, we connect the image file paths and labels to the generators.  \nThe generator will:\n- Load each image from the directory\n- Resize it to 224×224\n- Apply augmentation (for training)\n- Yield batches automatically during training\n","metadata":{}},{"cell_type":"code","source":"print(\"Creating data generators...\")\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=df_train,\n    directory=PNG_DIR,\n    x_col='filename',\n    y_col='Target',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='binary',\n    validate_filenames=False\n)\n\nvalidation_generator = val_datagen.flow_from_dataframe(\n    dataframe=df_val,\n    directory=PNG_DIR,\n    x_col='filename',\n    y_col='Target',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='binary',\n    shuffle=False,\n    validate_filenames=False\n)\n\nprint(\"Data generators created successfully.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:25:26.577034Z","iopub.execute_input":"2025-10-08T12:25:26.577608Z","iopub.status.idle":"2025-10-08T12:25:26.651300Z","shell.execute_reply.started":"2025-10-08T12:25:26.577583Z","shell.execute_reply":"2025-10-08T12:25:26.650714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 5: Calculate Class Weights\nIf one class (e.g., *normal*) has far more samples than the other (e.g., *pneumonia*),\nwe compute **class weights** so the model pays equal attention to both.\n","metadata":{}},{"cell_type":"code","source":"class_weights = compute_class_weight(\n    'balanced',\n    classes=np.unique(train_generator.classes),\n    y=train_generator.classes\n)\nclass_weight_dict = dict(enumerate(class_weights))\n\nprint(f\"\\nCalculated Class Weights to handle imbalance: {class_weight_dict}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:25:31.837137Z","iopub.execute_input":"2025-10-08T12:25:31.837683Z","iopub.status.idle":"2025-10-08T12:25:31.847453Z","shell.execute_reply.started":"2025-10-08T12:25:31.837663Z","shell.execute_reply":"2025-10-08T12:25:31.846901Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Building the Model\n\nWe will use **transfer learning**, a powerful technique where we take a pre-trained model (in this case, **ResNet50**, which was trained on millions of images from ImageNet) and adapt it for our specific task.\n\nOur strategy is a two-phase training process:\n1.  **Feature Extraction**: We will \"freeze\" the convolutional base of ResNet50 and only train a new, custom classifier head that we add on top. This allows our model to learn to classify X-rays using the powerful features ResNet50 already knows how to detect.\n2.  **Fine-Tuning**: After the initial training, we will \"unfreeze\" some of the top layers of the ResNet50 base and continue training with a very low learning rate. This fine-tunes the pre-trained features to be more specific to chest X-rays.","metadata":{}},{"cell_type":"markdown","source":"## Step 7.1.1: Define and Compile the Basic CNN Model\n\nWe start with a simple Convolutional Neural Network (CNN) architecture to classify chest X-ray images as **Normal** or **Pneumonia**.  \nThis model includes:\n- Two convolution + pooling layers for feature extraction  \n- A dense layer for learning complex patterns  \n- A dropout layer to prevent overfitting  \n- A sigmoid output layer for binary classification","metadata":{}},{"cell_type":"code","source":"\n\n# --- Define a Basic CNN ---\nmodel = Sequential([\n    Conv2D(32, (3,3), activation='relu', input_shape=(224, 224, 3)),\n    MaxPooling2D(2,2),\n\n    Conv2D(64, (3,3), activation='relu'),\n    MaxPooling2D(2,2),\n\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')  # Binary classification: Pneumonia (1) vs Normal (0)\n])\n\n# --- Compile the model ---\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\n# --- Summary ---\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T12:25:37.730652Z","iopub.execute_input":"2025-10-08T12:25:37.731270Z","iopub.status.idle":"2025-10-08T12:25:39.879548Z","shell.execute_reply.started":"2025-10-08T12:25:37.731245Z","shell.execute_reply":"2025-10-08T12:25:39.879043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 7.1.2: Train the Model\n\nWe train the CNN on the training data and validate it on unseen images.  \nClass weights are applied to handle the imbalance between Normal and Pneumonia samples.  \nStart with a small number of epochs (e.g., 10) and increase later if results improve.","metadata":{}},{"cell_type":"code","source":"# --- Train the Basic CNN ---\nEPOCHS = 10  # start small, increase later if working\nstart=time.time()\nhistory = model.fit(\n    train_generator,\n    validation_data=validation_generator,\n    epochs=EPOCHS,\n    class_weight=class_weight_dict,  # handle imbalance\n    verbose=1\n)\nend=time.time()\nelapsed=end-start\nprint(f\"\\n✅ Metadata extraction complete in {elapsed:.2f} seconds.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 7.1.3: Save the Trained Model\n\nAfter training, we save the model as an `.h5` file so we can load it later without retraining.  \nThis saves both the **model architecture** and **weights**.","metadata":{}},{"cell_type":"code","source":"model.save('/kaggle/working/basic_cnn_model.h5')\nprint(\"✅ Model saved successfully!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 7.1.4: Visualize Training Performance\n\nWe plot the training and validation **accuracy** and **loss** over epochs to evaluate model performance.  \nThese plots help us identify:\n- Overfitting (training accuracy much higher than validation)\n- Underfitting (both accuracies low)\n- Convergence (loss decreasing steadily)\n  ","metadata":{}},{"cell_type":"code","source":"\n# --- Plot accuracy and loss ---\nplt.figure(figsize=(12,5))\n\n# Accuracy plot\nplt.subplot(1,2,1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Val Accuracy')\nplt.legend()\nplt.title('Accuracy')\n\n# Loss plot\nplt.subplot(1,2,2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.legend()\nplt.title('Loss')\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. Build the Model ---\n# Load ResNet50 without its top classification layer\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(IMG_SIZE, IMG_SIZE, 3))\n\n# Freeze the base model layers\nbase_model.trainable = False\n\n# Add our custom classifier on top\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.5)(x)\npredictions = Dense(1, activation='sigmoid')(x)\n\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Feature Extraction Training\nprint(\"\\\\n--- Phase 1: Training the Classifier Head ---\")\nmodel.compile(optimizer=Adam(learning_rate=1e-4), loss='binary_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(\n    train_generator,\n    epochs=5, # A few epochs are enough for the head\n    validation_data=validation_generator,\n    class_weight=class_weight_dict,\n    verbose=1\n)\n\n# Phase 2: Fine-Tuning\nprint(\"\\\\n--- Phase 2: Fine-Tuning the Top Layers ---\")\nbase_model.trainable = True\n\n# unfreeze the top 20% of the layers\nfine_tune_at = int(len(base_model.layers) * 0.8)\nfor layer in base_model.layers[:fine_tune_at]:\n    layer.trainable = False\n    \nmodel.compile(optimizer=Adam(learning_rate=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Use callbacks to stop training when performance stops improving\ncallbacks = [\n    ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2, verbose=1),\n    EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True, verbose=1)\n]\n\nhistory_fine_tune = model.fit(\n    train_generator,\n    epochs=20, # Train for more epochs, but EarlyStopping will find the best one\n    initial_epoch=history.epoch[-1] + 1,\n    validation_data=validation_generator,\n    class_weight=class_weight_dict,\n    callbacks=callbacks,\n    verbose=1\n)\n\n# Plot Training History \ndef plot_history(history, history_fine, initial_epochs=5):\n    acc = history.history['accuracy'] + history_fine.history['accuracy']\n    val_acc = history.history['val_accuracy'] + history_fine.history['val_accuracy']\n    loss = history.history['loss'] + history_fine.history['loss']\n    val_loss = history.history['val_loss'] + history_fine.history['val_loss']\n\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(acc, label='Training Accuracy')\n    plt.plot(val_acc, label='Validation Accuracy')\n    plt.plot([initial_epochs-1, initial_epochs-1], plt.ylim(), label='Start Fine-Tuning')\n    plt.legend(loc='lower right')\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epoch')\n\n    plt.subplot(1, 2, 2)\n    plt.plot(loss, label='Training Loss')\n    plt.plot(val_loss, label='Validation Loss')\n    plt.plot([initial_epochs-1, initial_epochs-1], plt.ylim(), label='Start Fine-Tuning')\n    plt.legend(loc='upper right')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epoch')\n    \n    plt.show()\n\nplot_history(history, history_fine_tune)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# NLP Radiology Reports \n\n\n\nIn this section, we will:\n- Import required libraries for text processing and visualization.  \n- Define the base path and subfolders for the curated CXR report dataset.  \n- Prepare helper functions to detect text columns, clean text, and generate word clouds.","metadata":{}},{"cell_type":"code","source":"# Base path and subfolders\nnlp_base_path = \"/kaggle/input/curated-cxr-report-generation-dataset\"\nnlp_folders = [\"NLP_aug_datasets\", \"Cleanses csv tfrecords\"]\n\n# Detect text column by average string length\ndef detect_text_column(df):\n    text_cols = [c for c in df.columns if df[c].dtype == object]\n    if len(text_cols) == 0:\n        return None\n    col_lengths = df[text_cols].astype(str).applymap(len).mean()\n    return col_lengths.idxmax()\n\n# Clean text column (lowercase, strip, normalize spaces)\ndef clean_text_column(df, col_name):\n    df[col_name] = df[col_name].astype(str).str.lower()\n    df[col_name] = df[col_name].str.replace(r'\\s+', ' ', regex=True).str.strip()\n    return df\n\n# Word cloud generator\ndef create_word_cloud(text_data, title):\n    if len(text_data) == 0:\n        print(\"No text data found.\")\n        return\n    wc = WordCloud(width=800, height=400, background_color='white').generate(' '.join(text_data))\n    plt.figure(figsize=(12,6))\n    plt.imshow(wc, interpolation='bilinear')\n    plt.axis('off')\n    plt.title(title, fontsize=16)\n    plt.show()\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# NLP Dataset Exploration\n\nNow we will:\n- Iterate through the subfolders (`NLP_aug_datasets`, `Cleanses csv tfrecords`).  \n- Detect and clean the main text column in each CSV.  \n- Show sample cleaned text.  \n- Generate word clouds to visualize the most frequent terms in radiology reports.","metadata":{}},{"cell_type":"code","source":"for folder in nlp_folders:\n    folder_path = os.path.join(nlp_base_path, folder)\n    csv_files = [f for f in os.listdir(folder_path) if f.endswith('.csv')]\n    print(f\"\\nFolder: {folder} contains CSVs: {csv_files}\")\n\n    for csv_file in csv_files:\n        print(f\"\\nProcessing {csv_file} ...\")\n        df_nlp = pd.read_csv(os.path.join(folder_path, csv_file))\n        print(f\"Dataset shape: {df_nlp.shape}\")\n        \n        # Detect text column\n        text_col = detect_text_column(df_nlp)\n        if text_col is None:\n            print(\"No text column detected. Skipping this file.\")\n            continue\n        \n        print(f\"Detected text column: {text_col}\")\n        df_nlp = clean_text_column(df_nlp, text_col)\n        \n        # Show sample cleaned text\n        print(f\"Sample cleaned text:\\n{df_nlp[text_col].head(3).tolist()}\")\n        \n        # Generate word cloud\n        create_word_cloud(df_nlp[text_col].dropna().astype(str).tolist(), f\"Word Cloud for {csv_file}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  NLP Section Completed\n\nWe have:\n- Processed multiple CSVs from the curated CXR report dataset.  \n- Automatically detected and cleaned the main text column.  \n- Visualized common terms in radiology reports using word clouds.  \n\nThis gives us an overview of the language patterns in the dataset.","metadata":{}}]}