{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ── 1 · Imports & Configuration ────────────────────────────\nimport os\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport pydicom\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    classification_report, confusion_matrix,\n    accuracy_score, precision_score, recall_score,\n    f1_score, roc_auc_score\n)\nfrom sklearn.utils.class_weight import compute_class_weight\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import EfficientNetB2\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\nwarnings.filterwarnings(\"ignore\")\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\nIMG_SIZE   = 224\nBATCH_SIZE = 16\nEPOCHS     = 5\nAUTOTUNE   = tf.data.AUTOTUNE\n\nWORKDIR = \"/kaggle/working/rsna_multichannel\"\nos.makedirs(WORKDIR, exist_ok=True)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:18.258213Z","iopub.execute_input":"2026-05-06T21:59:18.258768Z","iopub.status.idle":"2026-05-06T21:59:18.265019Z","shell.execute_reply.started":"2026-05-06T21:59:18.258742Z","shell.execute_reply":"2026-05-06T21:59:18.264185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 2 · RSNA Dataset ──────────────────────────────────────\nRSNA_DIR = \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\"\nIMAGE_DIR = os.path.join(RSNA_DIR, \"stage_2_train_images\")\nLABELS_CSV = os.path.join(RSNA_DIR, \"stage_2_train_labels.csv\")\n\ndf = pd.read_csv(LABELS_CSV)\n\ndf[\"label\"] = df[\"Target\"]\ndf = df.drop_duplicates(subset=[\"patientId\"])\n\ndf[\"label_name\"] = df[\"label\"].map({\n    0: \"Normal\",\n    1: \"Viral_pneumonia\"\n})\n\ndf[\"image_path\"] = df[\"patientId\"].apply(\n    lambda x: os.path.join(IMAGE_DIR, f\"{x}.dcm\")\n)\n\nCLASS_NAMES = [\"Normal\", \"Viral_pneumonia\"]\n\nprint(f\"\\nTotal images: {len(df)}\")\nprint(df[\"label_name\"].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:18.266625Z","iopub.execute_input":"2026-05-06T21:59:18.267115Z","iopub.status.idle":"2026-05-06T21:59:18.376547Z","shell.execute_reply.started":"2026-05-06T21:59:18.267091Z","shell.execute_reply":"2026-05-06T21:59:18.375742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 3 · Split 80/10/10 ────────────────────────────────────\ntrain_df, temp_df = train_test_split(\n    df, test_size=0.20, stratify=df[\"label\"], random_state=SEED\n)\nval_df, test_df = train_test_split(\n    temp_df, test_size=0.50, stratify=temp_df[\"label\"], random_state=SEED\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:18.377517Z","iopub.execute_input":"2026-05-06T21:59:18.377835Z","iopub.status.idle":"2026-05-06T21:59:18.402340Z","shell.execute_reply.started":"2026-05-06T21:59:18.377805Z","shell.execute_reply":"2026-05-06T21:59:18.401839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Class weights ─────────────────────────────────────────\nweights = compute_class_weight(\n    \"balanced\",\n    classes=np.unique(train_df[\"label\"]),\n    y=train_df[\"label\"]\n)\nclass_weight_dict = dict(enumerate(weights))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:18.403026Z","iopub.execute_input":"2026-05-06T21:59:18.403205Z","iopub.status.idle":"2026-05-06T21:59:18.413598Z","shell.execute_reply.started":"2026-05-06T21:59:18.403188Z","shell.execute_reply":"2026-05-06T21:59:18.412959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 4 · Preprocessing ─────────────────────────────────────\ndef gamma_correction(img, gamma=0.5):\n    inv = 1.0 / gamma\n    lut = np.array([(i/255.0)**inv * 255 for i in range(256)], dtype=np.uint8)\n    return cv2.LUT(img, lut)\n\n\ndef sobel_magnitude(img):\n    sx = cv2.Sobel(img, cv2.CV_32F, 1, 0)\n    sy = cv2.Sobel(img, cv2.CV_32F, 0, 1)\n    mag = np.sqrt(sx**2 + sy**2)\n    p99 = np.percentile(mag, 99)\n    return np.clip(mag/(p99+1e-7), 0, 1).astype(np.float32)\n\n\ndef multi_channel_fn(path):\n    path = path.numpy().decode()\n\n    try:\n        dcm = pydicom.dcmread(path)\n        img = dcm.pixel_array\n    except:\n        return np.zeros((IMG_SIZE, IMG_SIZE, 3), dtype=np.float32)\n\n    img = img.astype(np.float32)\n    img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n    img = (img * 255).astype(np.uint8)\n\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    img = cv2.bilateralFilter(img, 9, 75, 75)\n\n    clahe = cv2.createCLAHE(2.0, (8,8))\n    ch1 = clahe.apply(img).astype(np.float32) / 255.0\n    ch2 = sobel_magnitude(img)\n    ch3 = gamma_correction(img, 0.5).astype(np.float32) / 255.0\n\n    return np.stack([ch1, ch2, ch3], axis=-1)\n\n\ndef process_data(path, label):\n    img = tf.py_function(multi_channel_fn, [path], tf.float32)\n    img.set_shape([IMG_SIZE, IMG_SIZE, 3])\n    img = preprocess_input(img * 255.0)\n    return img, tf.expand_dims(tf.cast(label, tf.float32), -1)\n\n\naugmentation = tf.keras.Sequential([\n    layers.RandomFlip(\"horizontal\"),\n    layers.RandomContrast(0.1),\n])\n\n\ndef process_train(path, label):\n    img, lbl = process_data(path, label)\n    return augmentation(img, training=True), lbl\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:18.414977Z","iopub.execute_input":"2026-05-06T21:59:18.415474Z","iopub.status.idle":"2026-05-06T21:59:19.918534Z","shell.execute_reply.started":"2026-05-06T21:59:18.415426Z","shell.execute_reply":"2026-05-06T21:59:19.917652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 5 · tf.data ───────────────────────────────────────────\ntrain_ds = tf.data.Dataset.from_tensor_slices(\n    (train_df[\"image_path\"], train_df[\"label\"])\n).shuffle(1000).map(process_train, num_parallel_calls=AUTOTUNE).batch(BATCH_SIZE).prefetch(AUTOTUNE)\n\nval_ds = tf.data.Dataset.from_tensor_slices(\n    (val_df[\"image_path\"], val_df[\"label\"])\n).map(process_data, num_parallel_calls=AUTOTUNE).batch(BATCH_SIZE).prefetch(AUTOTUNE)\n\ntest_ds = tf.data.Dataset.from_tensor_slices(\n    (test_df[\"image_path\"], test_df[\"label\"])\n).map(process_data, num_parallel_calls=AUTOTUNE).batch(BATCH_SIZE).prefetch(AUTOTUNE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:19.919379Z","iopub.execute_input":"2026-05-06T21:59:19.919632Z","iopub.status.idle":"2026-05-06T21:59:21.653785Z","shell.execute_reply.started":"2026-05-06T21:59:19.919612Z","shell.execute_reply":"2026-05-06T21:59:21.653159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 6 · Model ─────────────────────────────────────────────\nbase = EfficientNetB2(weights=\"imagenet\", include_top=False, input_shape=(IMG_SIZE,IMG_SIZE,3))\nbase.trainable = False\n\nx = base.output\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.BatchNormalization()(x)\nx = layers.Dropout(0.4)(x)\nx = layers.Dense(128, activation=\"relu\")(x)\nout = layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = Model(base.input, out)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-3),\n    loss=\"binary_crossentropy\",\n    metrics=[\"accuracy\", tf.keras.metrics.AUC(name=\"auc\")]\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:21.655293Z","iopub.execute_input":"2026-05-06T21:59:21.655598Z","iopub.status.idle":"2026-05-06T21:59:27.007037Z","shell.execute_reply.started":"2026-05-06T21:59:21.655574Z","shell.execute_reply":"2026-05-06T21:59:27.006447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 7 · Training ──────────────────────────────────────────\ncallbacks = [\n    ModelCheckpoint(f\"{WORKDIR}/best.keras\", save_best_only=True),\n    EarlyStopping(patience=3, restore_best_weights=True),\n    ReduceLROnPlateau(patience=2)\n]\n\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    class_weight=class_weight_dict,\n    callbacks=callbacks\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:59:27.008091Z","iopub.execute_input":"2026-05-06T21:59:27.008392Z","iopub.status.idle":"2026-05-06T22:16:53.129960Z","shell.execute_reply.started":"2026-05-06T21:59:27.008363Z","shell.execute_reply":"2026-05-06T22:16:53.129284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 8 · Evaluation ────────────────────────────────────────\ny_true = test_df[\"label\"].values\ny_prob = model.predict(test_ds).ravel()\ny_pred = (y_prob >= 0.5).astype(int)\n\nprint(\"\\nRESULTS\")\nprint(\"Accuracy :\", accuracy_score(y_true, y_pred))\nprint(\"Precision:\", precision_score(y_true, y_pred))\nprint(\"Recall   :\", recall_score(y_true, y_pred))\nprint(\"F1       :\", f1_score(y_true, y_pred))\nprint(\"AUC      :\", roc_auc_score(y_true, y_prob))\n\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_true, y_pred, target_names=CLASS_NAMES))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T22:16:53.131863Z","iopub.execute_input":"2026-05-06T22:16:53.132060Z","iopub.status.idle":"2026-05-06T22:17:45.134369Z","shell.execute_reply.started":"2026-05-06T22:16:53.132042Z","shell.execute_reply":"2026-05-06T22:17:45.133560Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Confusion Matrix ──────────────────────────────────────\ncm = confusion_matrix(y_true, y_pred)\n\nplt.figure(figsize=(6,5))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\",\n            xticklabels=CLASS_NAMES,\n            yticklabels=CLASS_NAMES)\nplt.title(\"Confusion Matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T22:17:45.135254Z","iopub.execute_input":"2026-05-06T22:17:45.135596Z","iopub.status.idle":"2026-05-06T22:17:45.287778Z","shell.execute_reply.started":"2026-05-06T22:17:45.135570Z","shell.execute_reply":"2026-05-06T22:17:45.286793Z"}},"outputs":[],"execution_count":null}]}