{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"9ee0350a","cell_type":"markdown","source":"# RSNA Pneumonia Detection Challenge — v2 Precision/F1 Optimized\n\nThis notebook is tuned from your latest results:\n\n- Accuracy: 0.7567\n- Precision: 0.4814\n- Recall: 0.7110\n- F1-score: 0.5741\n- ROC AUC: 0.8174\n\n## Goal of this version\nKeep recall reasonably high while improving:\n- **precision**\n- **F1-score**\n- **AUC**\n\n## Main changes\n- CLAHE preprocessing for chest X-rays\n- EfficientNetB0 transfer learning\n- milder class-weighting\n- binary crossentropy instead of focal loss\n- longer fine-tuning\n- tighter threshold search optimized for F1\n- precision/recall/threshold plots\n","metadata":{}},{"id":"f7379977","cell_type":"code","source":"import os\nimport random\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport pydicom\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    roc_auc_score,\n    classification_report,\n    confusion_matrix,\n    precision_recall_curve,\n)\n\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\n\nprint(\"TensorFlow version:\", tf.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:14:47.528815Z","iopub.execute_input":"2026-03-23T06:14:47.529543Z","iopub.status.idle":"2026-03-23T06:15:10.337018Z","shell.execute_reply.started":"2026-03-23T06:14:47.529512Z","shell.execute_reply":"2026-03-23T06:15:10.336285Z"}},"outputs":[],"execution_count":null},{"id":"69d3387e","cell_type":"code","source":"# =========================\n# Config\n# =========================\nDATASET_ROOT = Path(\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\")\nTRAIN_IMAGE_DIR = DATASET_ROOT / \"stage_2_train_images\"\nLABELS_CSV = DATASET_ROOT / \"stage_2_train_labels.csv\"\n\nIMG_SIZE = 224\nSEED = 42\nMAX_IMAGES = 12000\nBATCH_SIZE = 16\nEPOCHS_STAGE1 = 8\nEPOCHS_STAGE2 = 12\n\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\nif not TRAIN_IMAGE_DIR.exists():\n    raise FileNotFoundError(f\"Missing folder: {TRAIN_IMAGE_DIR}\")\n\nif not LABELS_CSV.exists():\n    raise FileNotFoundError(f\"Missing file: {LABELS_CSV}\")\n\nprint(\"Dataset root:\", DATASET_ROOT)\nprint(\"Train images:\", TRAIN_IMAGE_DIR)\nprint(\"Labels file:\", LABELS_CSV)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:15:10.338353Z","iopub.execute_input":"2026-03-23T06:15:10.338895Z","iopub.status.idle":"2026-03-23T06:15:10.351119Z","shell.execute_reply.started":"2026-03-23T06:15:10.338846Z","shell.execute_reply":"2026-03-23T06:15:10.350461Z"}},"outputs":[],"execution_count":null},{"id":"40a471b9","cell_type":"code","source":"# =========================\n# Load binary labels\n# =========================\nlabels_df = pd.read_csv(LABELS_CSV)\n\nbinary_df = (\n    labels_df.groupby(\"patientId\", as_index=False)[\"Target\"]\n    .max()\n    .rename(columns={\"Target\": \"label\"})\n)\nbinary_df[\"label\"] = binary_df[\"label\"].astype(int)\n\nprint(\"Label distribution:\")\ndisplay(binary_df[\"label\"].value_counts())\ndisplay(binary_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:15:10.351902Z","iopub.execute_input":"2026-03-23T06:15:10.352125Z","iopub.status.idle":"2026-03-23T06:15:10.460026Z","shell.execute_reply.started":"2026-03-23T06:15:10.352101Z","shell.execute_reply":"2026-03-23T06:15:10.459455Z"}},"outputs":[],"execution_count":null},{"id":"9b75af1f","cell_type":"code","source":"# =========================\n# Match labels to DICOM files\n# =========================\nall_dcm_files = sorted(TRAIN_IMAGE_DIR.glob(\"*.dcm\"))\nimage_map = {fp.stem: fp for fp in all_dcm_files}\n\nbinary_df[\"path\"] = binary_df[\"patientId\"].map(lambda x: str(image_map[x]) if x in image_map else None)\nbinary_df = binary_df.dropna(subset=[\"path\"]).reset_index(drop=True)\n\nprint(\"Matched images:\", len(binary_df))\n\nif len(binary_df) > MAX_IMAGES:\n    binary_df = binary_df.sample(MAX_IMAGES, random_state=SEED).reset_index(drop=True)\n    print(\"Using subset:\", len(binary_df))\n\ndisplay(binary_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:15:10.461784Z","iopub.execute_input":"2026-03-23T06:15:10.462053Z","iopub.status.idle":"2026-03-23T06:15:12.000970Z","shell.execute_reply.started":"2026-03-23T06:15:10.462028Z","shell.execute_reply":"2026-03-23T06:15:12.000327Z"}},"outputs":[],"execution_count":null},{"id":"eb3780b5","cell_type":"code","source":"# =========================\n# DICOM loader with CLAHE\n# =========================\ndef load_dicom_rgb(path: str, img_size: int = 224) -> np.ndarray:\n    dcm = pydicom.dcmread(path)\n    img = dcm.pixel_array.astype(np.float32)\n\n    if getattr(dcm, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = img.max() - img\n\n    img = img - img.min()\n    if img.max() > 0:\n        img = img / img.max()\n\n    img = (img * 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    img = clahe.apply(img)\n\n    img = cv2.resize(img, (img_size, img_size))\n    img = np.stack([img, img, img], axis=-1).astype(np.float32)\n    return img\n\npos_row = binary_df[binary_df[\"label\"] == 1].head(1)\nneg_row = binary_df[binary_df[\"label\"] == 0].head(1)\n\nplt.figure(figsize=(8, 4))\nif len(pos_row):\n    plt.subplot(1, 2, 1)\n    plt.imshow(load_dicom_rgb(pos_row.iloc[0][\"path\"], IMG_SIZE).astype(np.uint8))\n    plt.title(\"Positive / Pneumonia\")\n    plt.axis(\"off\")\n\nif len(neg_row):\n    plt.subplot(1, 2, 2)\n    plt.imshow(load_dicom_rgb(neg_row.iloc[0][\"path\"], IMG_SIZE).astype(np.uint8))\n    plt.title(\"Negative / No Pneumonia\")\n    plt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:15:12.001989Z","iopub.execute_input":"2026-03-23T06:15:12.002243Z","iopub.status.idle":"2026-03-23T06:15:12.343953Z","shell.execute_reply.started":"2026-03-23T06:15:12.002221Z","shell.execute_reply":"2026-03-23T06:15:12.343052Z"}},"outputs":[],"execution_count":null},{"id":"1657d6e3","cell_type":"code","source":"# =========================\n# Load images\n# =========================\nimages = []\nlabels = []\n\nfor _, row in binary_df.iterrows():\n    try:\n        img = load_dicom_rgb(row[\"path\"], IMG_SIZE)\n        images.append(img)\n        labels.append(int(row[\"label\"]))\n    except Exception as e:\n        print(\"Skipped:\", row[\"path\"], e)\n\nX = np.array(images, dtype=np.float32)\ny = np.array(labels, dtype=np.int32)\n\nprint(\"Loaded X shape:\", X.shape)\nprint(\"Loaded y shape:\", y.shape)\nprint(\"Positive ratio:\", y.mean())\n\nX = preprocess_input(X)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:15:12.344953Z","iopub.execute_input":"2026-03-23T06:15:12.345261Z","iopub.status.idle":"2026-03-23T06:19:00.163083Z","shell.execute_reply.started":"2026-03-23T06:15:12.345219Z","shell.execute_reply":"2026-03-23T06:19:00.162237Z"}},"outputs":[],"execution_count":null},{"id":"c28827a3","cell_type":"code","source":"# =========================\n# Split\n# =========================\nx_train, x_temp, y_train, y_temp = train_test_split(\n    X, y,\n    test_size=0.30,\n    random_state=SEED,\n    stratify=y\n)\n\nx_val, x_test, y_val, y_test = train_test_split(\n    x_temp, y_temp,\n    test_size=0.50,\n    random_state=SEED,\n    stratify=y_temp\n)\n\nprint(\"Train:\", x_train.shape, y_train.shape)\nprint(\"Val  :\", x_val.shape, y_val.shape)\nprint(\"Test :\", x_test.shape, y_test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:19:00.164331Z","iopub.execute_input":"2026-03-23T06:19:00.164720Z","iopub.status.idle":"2026-03-23T06:19:02.861877Z","shell.execute_reply.started":"2026-03-23T06:19:00.164691Z","shell.execute_reply":"2026-03-23T06:19:02.861043Z"}},"outputs":[],"execution_count":null},{"id":"999e74b1","cell_type":"code","source":"# =========================\n# Milder class weights\n# =========================\nbase_weights = class_weight.compute_class_weight(\n    class_weight=\"balanced\",\n    classes=np.unique(y_train),\n    y=y_train\n)\nclass_weights = dict(enumerate(base_weights))\nclass_weights[1] = float(class_weights[1] * 0.85)\n\nprint(\"Base class weights:\", dict(enumerate(base_weights)))\nprint(\"Adjusted class weights:\", class_weights)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:19:02.863283Z","iopub.execute_input":"2026-03-23T06:19:02.863856Z","iopub.status.idle":"2026-03-23T06:19:02.874809Z","shell.execute_reply.started":"2026-03-23T06:19:02.863827Z","shell.execute_reply":"2026-03-23T06:19:02.874000Z"}},"outputs":[],"execution_count":null},{"id":"a64789b1","cell_type":"code","source":"# =========================\n# Data augmentation\n# =========================\ntrain_datagen = tf.keras.preprocessing.image.ImageDataGenerator(\n    rotation_range=8,\n    zoom_range=0.08,\n    width_shift_range=0.04,\n    height_shift_range=0.04,\n    horizontal_flip=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:19:02.875787Z","iopub.execute_input":"2026-03-23T06:19:02.876149Z","iopub.status.idle":"2026-03-23T06:19:02.880467Z","shell.execute_reply.started":"2026-03-23T06:19:02.876122Z","shell.execute_reply":"2026-03-23T06:19:02.879826Z"}},"outputs":[],"execution_count":null},{"id":"084402ef","cell_type":"code","source":"# =========================\n# Build model\n# =========================\nbase_model = EfficientNetB0(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\nbase_model.trainable = False\n\ninputs = tf.keras.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\nx = base_model(inputs, training=False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dense(256, activation=\"relu\")(x)\nx = tf.keras.layers.Dropout(0.4)(x)\nx = tf.keras.layers.Dense(64, activation=\"relu\")(x)\nx = tf.keras.layers.Dropout(0.3)(x)\noutputs = tf.keras.layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = tf.keras.Model(inputs, outputs)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-3),\n    loss=\"binary_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:19:02.883032Z","iopub.execute_input":"2026-03-23T06:19:02.883257Z","iopub.status.idle":"2026-03-23T06:19:07.240842Z","shell.execute_reply.started":"2026-03-23T06:19:02.883236Z","shell.execute_reply":"2026-03-23T06:19:07.240209Z"}},"outputs":[],"execution_count":null},{"id":"9a5f6f36","cell_type":"code","source":"# =========================\n# Train stage 1\n# =========================\ncallbacks_stage1 = [\n    ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=2,\n        verbose=1,\n        min_lr=1e-6\n    ),\n    EarlyStopping(\n        monitor=\"val_loss\",\n        patience=5,\n        restore_best_weights=True,\n        verbose=1\n    ),\n    ModelCheckpoint(\n        \"/kaggle/working/best_rsna_v2_stage1.keras\",\n        monitor=\"val_accuracy\",\n        save_best_only=True,\n        mode=\"max\",\n        verbose=1\n    )\n]\n\nhistory1 = model.fit(\n    train_datagen.flow(x_train, y_train, batch_size=BATCH_SIZE),\n    validation_data=(x_val, y_val),\n    epochs=EPOCHS_STAGE1,\n    class_weight=class_weights,\n    callbacks=callbacks_stage1,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:19:07.241714Z","iopub.execute_input":"2026-03-23T06:19:07.241989Z","iopub.status.idle":"2026-03-23T06:30:55.404593Z","shell.execute_reply.started":"2026-03-23T06:19:07.241966Z","shell.execute_reply":"2026-03-23T06:30:55.403600Z"}},"outputs":[],"execution_count":null},{"id":"6d7e2cf5","cell_type":"code","source":"# =========================\n# Fine-tune stage 2\n# =========================\nbase_model.trainable = True\n\nfor layer in base_model.layers[:100]:\n    layer.trainable = False\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-5),\n    loss=\"binary_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\ncallbacks_stage2 = [\n    ReduceLROnPlateau(\n        monitor=\"val_loss\",\n        factor=0.5,\n        patience=2,\n        verbose=1,\n        min_lr=1e-7\n    ),\n    EarlyStopping(\n        monitor=\"val_loss\",\n        patience=5,\n        restore_best_weights=True,\n        verbose=1\n    ),\n    ModelCheckpoint(\n        \"/kaggle/working/best_rsna_v2_finetuned.keras\",\n        monitor=\"val_accuracy\",\n        save_best_only=True,\n        mode=\"max\",\n        verbose=1\n    )\n]\n\nhistory2 = model.fit(\n    train_datagen.flow(x_train, y_train, batch_size=BATCH_SIZE),\n    validation_data=(x_val, y_val),\n    epochs=EPOCHS_STAGE2,\n    class_weight=class_weights,\n    callbacks=callbacks_stage2,\n    verbose=1\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:30:55.406031Z","iopub.execute_input":"2026-03-23T06:30:55.406322Z","iopub.status.idle":"2026-03-23T06:48:47.491903Z","shell.execute_reply.started":"2026-03-23T06:30:55.406295Z","shell.execute_reply":"2026-03-23T06:48:47.491138Z"}},"outputs":[],"execution_count":null},{"id":"006b69da","cell_type":"code","source":"# =========================\n# Curves\n# =========================\ntrain_acc = history1.history[\"accuracy\"] + history2.history[\"accuracy\"]\nval_acc = history1.history[\"val_accuracy\"] + history2.history[\"val_accuracy\"]\ntrain_loss = history1.history[\"loss\"] + history2.history[\"loss\"]\nval_loss = history1.history[\"val_loss\"] + history2.history[\"val_loss\"]\n\nplt.figure(figsize=(14, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(train_acc, label=\"train_acc\")\nplt.plot(val_acc, label=\"val_acc\")\nplt.title(\"Accuracy\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(train_loss, label=\"train_loss\")\nplt.plot(val_loss, label=\"val_loss\")\nplt.title(\"Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:48:47.495805Z","iopub.execute_input":"2026-03-23T06:48:47.496062Z","iopub.status.idle":"2026-03-23T06:48:47.835292Z","shell.execute_reply.started":"2026-03-23T06:48:47.496037Z","shell.execute_reply":"2026-03-23T06:48:47.834663Z"}},"outputs":[],"execution_count":null},{"id":"4b0ef6ff","cell_type":"code","source":"# =========================\n# Threshold tuning for best F1\n# =========================\nval_prob = model.predict(x_val, verbose=0).ravel()\n\nthresholds = np.linspace(0.20, 0.80, 121)\nrows = []\n\nfor t in thresholds:\n    val_pred = (val_prob >= t).astype(int)\n    rows.append({\n        \"threshold\": t,\n        \"precision\": precision_score(y_val, val_pred, zero_division=0),\n        \"recall\": recall_score(y_val, val_pred, zero_division=0),\n        \"f1\": f1_score(y_val, val_pred, zero_division=0)\n    })\n\nthr_df = pd.DataFrame(rows)\nbest_row = thr_df.loc[thr_df[\"f1\"].idxmax()]\nbest_threshold = float(best_row[\"threshold\"])\n\nprint(\"Best threshold:\", best_threshold)\nprint(best_row)\n\ndisplay(thr_df.sort_values(\"f1\", ascending=False).head(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:48:47.836305Z","iopub.execute_input":"2026-03-23T06:48:47.836659Z","iopub.status.idle":"2026-03-23T06:49:07.105520Z","shell.execute_reply.started":"2026-03-23T06:48:47.836600Z","shell.execute_reply":"2026-03-23T06:49:07.104612Z"}},"outputs":[],"execution_count":null},{"id":"c7159b7b","cell_type":"code","source":"# =========================\n# Threshold curves\n# =========================\nplt.figure(figsize=(10, 6))\nplt.plot(thr_df[\"threshold\"], thr_df[\"precision\"], label=\"precision\")\nplt.plot(thr_df[\"threshold\"], thr_df[\"recall\"], label=\"recall\")\nplt.plot(thr_df[\"threshold\"], thr_df[\"f1\"], label=\"f1\")\nplt.axvline(best_threshold, color=\"red\", linestyle=\"--\", label=f\"best={best_threshold:.2f}\")\nplt.xlabel(\"Threshold\")\nplt.ylabel(\"Score\")\nplt.title(\"Threshold tuning on validation set\")\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:49:07.106678Z","iopub.execute_input":"2026-03-23T06:49:07.107101Z","iopub.status.idle":"2026-03-23T06:49:07.281097Z","shell.execute_reply.started":"2026-03-23T06:49:07.107071Z","shell.execute_reply":"2026-03-23T06:49:07.280300Z"}},"outputs":[],"execution_count":null},{"id":"24248387","cell_type":"code","source":"# =========================\n# Final evaluation\n# =========================\ny_prob = model.predict(x_test, verbose=0).ravel()\ny_pred = (y_prob >= best_threshold).astype(int)\n\nacc = accuracy_score(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, zero_division=0)\nrecall = recall_score(y_test, y_pred, zero_division=0)\nf1 = f1_score(y_test, y_pred, zero_division=0)\nauc = roc_auc_score(y_test, y_prob)\n\nprint(f\"Accuracy : {acc:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"Recall   : {recall:.4f}\")\nprint(f\"F1-score : {f1:.4f}\")\nprint(f\"ROC AUC  : {auc:.4f}\")\n\nprint(\"\\nClassification report:\")\nprint(classification_report(\n    y_test,\n    y_pred,\n    target_names=[\"NO_PNEUMONIA\", \"PNEUMONIA\"],\n    zero_division=0\n))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:49:07.282198Z","iopub.execute_input":"2026-03-23T06:49:07.282658Z","iopub.status.idle":"2026-03-23T06:49:11.917325Z","shell.execute_reply.started":"2026-03-23T06:49:07.282591Z","shell.execute_reply":"2026-03-23T06:49:11.916584Z"}},"outputs":[],"execution_count":null},{"id":"92d24017","cell_type":"code","source":"# =========================\n# Confusion matrix\n# =========================\ncm = confusion_matrix(y_test, y_pred)\n\nplt.figure(figsize=(6, 5))\nsns.heatmap(\n    cm,\n    annot=True,\n    fmt=\"d\",\n    cmap=\"Blues\",\n    xticklabels=[\"NO_PNEUMONIA\", \"PNEUMONIA\"],\n    yticklabels=[\"NO_PNEUMONIA\", \"PNEUMONIA\"]\n)\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:49:11.918336Z","iopub.execute_input":"2026-03-23T06:49:11.918696Z","iopub.status.idle":"2026-03-23T06:49:12.055120Z","shell.execute_reply.started":"2026-03-23T06:49:11.918646Z","shell.execute_reply":"2026-03-23T06:49:12.054401Z"}},"outputs":[],"execution_count":null},{"id":"4c816870","cell_type":"code","source":"# =========================\n# Precision-recall curve\n# =========================\nprecisions, recalls, pr_thresholds = precision_recall_curve(y_test, y_prob)\n\nplt.figure(figsize=(7, 6))\nplt.plot(recalls, precisions)\nplt.xlabel(\"Recall\")\nplt.ylabel(\"Precision\")\nplt.title(\"Precision-Recall Curve\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:49:12.055986Z","iopub.execute_input":"2026-03-23T06:49:12.056283Z","iopub.status.idle":"2026-03-23T06:49:12.191210Z","shell.execute_reply.started":"2026-03-23T06:49:12.056259Z","shell.execute_reply":"2026-03-23T06:49:12.190587Z"}},"outputs":[],"execution_count":null},{"id":"a92cab6d","cell_type":"code","source":"# =========================\n# Probability histogram\n# =========================\nplt.figure(figsize=(8, 5))\nplt.hist(y_prob[y_test == 0], bins=40, alpha=0.6, label=\"NO_PNEUMONIA\")\nplt.hist(y_prob[y_test == 1], bins=40, alpha=0.6, label=\"PNEUMONIA\")\nplt.axvline(best_threshold, color=\"red\", linestyle=\"--\", label=f\"Threshold={best_threshold:.2f}\")\nplt.title(\"Predicted probability distribution\")\nplt.xlabel(\"Predicted probability\")\nplt.ylabel(\"Count\")\nplt.legend()\nplt.show()\n\nprint(\"Saved model:\")\nprint(\"/kaggle/working/best_rsna_v2_finetuned.keras\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T06:49:12.192183Z","iopub.execute_input":"2026-03-23T06:49:12.192542Z","iopub.status.idle":"2026-03-23T06:49:12.985454Z","shell.execute_reply.started":"2026-03-23T06:49:12.192506Z","shell.execute_reply":"2026-03-23T06:49:12.984600Z"}},"outputs":[],"execution_count":null}]}