{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ==============================================================================\n# EfficientNetB0 + Random Forest on RSNA Database\n# Multi-channel Preprocessing (Gray, CLAHE, Gaussian Blur)\n# True 80/10/10 split\n# ==============================================================================\n","metadata":{}},{"cell_type":"code","source":"# =========================\n# 1 · Locate RSNA Dataset\n# =========================\nimport os\nimport pathlib\nimport pandas as pd\nimport numpy as np\nimport cv2\nimport pydicom\n\nRSNA_DIR = pathlib.Path('/kaggle/input/competitions/rsna-pneumonia-detection-challenge')\nIMAGE_DIR = RSNA_DIR / 'stage_2_train_images'\nLABELS_CSV = RSNA_DIR / 'stage_2_train_labels.csv'\n\nprint('RSNA_DIR:', RSNA_DIR)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:13:08.594504Z","iopub.execute_input":"2026-05-06T21:13:08.595160Z","iopub.status.idle":"2026-05-06T21:13:10.801115Z","shell.execute_reply.started":"2026-05-06T21:13:08.595128Z","shell.execute_reply":"2026-05-06T21:13:10.800297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 2 · Prepare train/val/test (80/10/10)\n# =========================\nimport shutil\nimport random\nfrom sklearn.model_selection import train_test_split\n\nSTAGED = pathlib.Path('/kaggle/working/staged_rsna')\nrandom.seed(42)\n\nCLASSES = ['Normal', 'Viral Pneumonia']\n\nif STAGED.exists():\n    shutil.rmtree(STAGED)\n\nfor split in ('train', 'val', 'test'):\n    for cls in CLASSES:\n        (STAGED / split / cls).mkdir(parents=True, exist_ok=True)\n\ndf = pd.read_csv(LABELS_CSV)\n\ndf['label'] = df['Target']\ndf = df.drop_duplicates(subset=['patientId'])\n\ndf['class_name'] = df['label'].map({\n    0: 'Normal',\n    1: 'Viral Pneumonia'\n})\n\ntrain_df, temp_df = train_test_split(\n    df, test_size=0.2, stratify=df['label'], random_state=42\n)\n\nval_df, test_df = train_test_split(\n    temp_df, test_size=0.5, stratify=temp_df['label'], random_state=42\n)\n\n\ndef dicom_to_png(src_path, dst_path):\n    dcm = pydicom.dcmread(src_path)\n    img = dcm.pixel_array.astype(np.float32)\n\n    img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n    img = (img * 255).astype(np.uint8)\n\n    cv2.imwrite(str(dst_path), img)\n\n\ndef process_split(split_df, split_name):\n    for _, row in split_df.iterrows():\n        patient_id = row['patientId']\n        cls = row['class_name']\n\n        src = IMAGE_DIR / f\"{patient_id}.dcm\"\n        dst = STAGED / split_name / cls / f\"{patient_id}.png\"\n\n        dicom_to_png(src, dst)\n\n\nprint(\"Processing TRAIN...\")\nprocess_split(train_df, 'train')\n\nprint(\"Processing VAL...\")\nprocess_split(val_df, 'val')\n\nprint(\"Processing TEST...\")\nprocess_split(test_df, 'test')\n\nfor split in ('train', 'val', 'test'):\n    for cls in CLASSES:\n        n = len(list((STAGED / split / cls).glob('*')))\n        print(f'{split}/{cls}: {n} images')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:13:10.802477Z","iopub.execute_input":"2026-05-06T21:13:10.802943Z","iopub.status.idle":"2026-05-06T21:29:51.565157Z","shell.execute_reply.started":"2026-05-06T21:13:10.802919Z","shell.execute_reply":"2026-05-06T21:29:51.564428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 3 · Imports\n# =========================\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score,\n    f1_score, roc_auc_score, confusion_matrix,\n    classification_report\n)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nprint('TensorFlow:', tf.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:29:51.566141Z","iopub.execute_input":"2026-05-06T21:29:51.566429Z","iopub.status.idle":"2026-05-06T21:30:17.841427Z","shell.execute_reply.started":"2026-05-06T21:29:51.566405Z","shell.execute_reply":"2026-05-06T21:30:17.840468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 4 · Data Generators\n# =========================\nTRAIN_DIR = str(STAGED / 'train')\nVAL_DIR   = str(STAGED / 'val')\nTEST_DIR  = str(STAGED / 'test')\n\nBATCH_SIZE = 32\nIMG_SIZE = (224, 224)\n\n\ndef create_multichannel_xray(image):\n    gray = cv2.cvtColor(image.astype(np.float32), cv2.COLOR_RGB2GRAY)\n    gray = cv2.normalize(gray, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n\n    ch1 = gray\n    ch2 = cv2.createCLAHE(2.0, (8,8)).apply(gray)\n    ch3 = cv2.GaussianBlur(gray, (5,5), 0)\n\n    return np.stack([ch1, ch2, ch3], axis=-1).astype(np.float32)\n\n\ndatagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    preprocessing_function=create_multichannel_xray\n)\n\ntrain_gen = datagen.flow_from_directory(TRAIN_DIR, target_size=IMG_SIZE, batch_size=BATCH_SIZE, class_mode='sparse', shuffle=False)\nval_gen   = datagen.flow_from_directory(VAL_DIR, target_size=IMG_SIZE, batch_size=BATCH_SIZE, class_mode='sparse', shuffle=False)\ntest_gen  = datagen.flow_from_directory(TEST_DIR, target_size=IMG_SIZE, batch_size=BATCH_SIZE, class_mode='sparse', shuffle=False)\n\nCLASS_NAMES = list(train_gen.class_indices.keys())\nprint('Classes:', CLASS_NAMES)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:30:17.843329Z","iopub.execute_input":"2026-05-06T21:30:17.843806Z","iopub.status.idle":"2026-05-06T21:30:18.073303Z","shell.execute_reply.started":"2026-05-06T21:30:17.843733Z","shell.execute_reply":"2026-05-06T21:30:18.072470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 5 · EfficientNetB0\n# =========================\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(224,224,3))\nbase_model.trainable = False\n\n\ndef extract_features(generator):\n    generator.reset()\n    feats = base_model.predict(generator, verbose=1)\n    return feats.reshape(feats.shape[0], -1)\n\n\nprint(\"Extracting features...\")\nX_train = extract_features(train_gen)\ny_train = train_gen.classes\n\nX_val = extract_features(val_gen)\ny_val = val_gen.classes\n\nX_test = extract_features(test_gen)\ny_test = test_gen.classes\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:30:18.074335Z","iopub.execute_input":"2026-05-06T21:30:18.074647Z","iopub.status.idle":"2026-05-06T21:36:48.917483Z","shell.execute_reply.started":"2026-05-06T21:30:18.074613Z","shell.execute_reply":"2026-05-06T21:36:48.916830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 6 · Random Forest\n# =========================\nrf = RandomForestClassifier(n_estimators=100, random_state=42, n_jobs=-1)\nrf.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:36:48.918642Z","iopub.execute_input":"2026-05-06T21:36:48.918980Z","iopub.status.idle":"2026-05-06T21:47:31.968104Z","shell.execute_reply.started":"2026-05-06T21:36:48.918942Z","shell.execute_reply":"2026-05-06T21:47:31.967421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 7 · Metrics\n# =========================\ny_pred = rf.predict(X_test)\ny_prob = rf.predict_proba(X_test)\n\naccuracy  = accuracy_score(y_test, y_pred)\nprecision = precision_score(y_test, y_pred, average='weighted', zero_division=0)\nrecall    = recall_score(y_test, y_pred, average='weighted', zero_division=0)\nf1        = f1_score(y_test, y_pred, average='weighted', zero_division=0)\n\nauc = roc_auc_score(y_test, y_prob[:, 1])\n\nprint(f'Accuracy  : {accuracy*100:.2f}%')\nprint(f'Precision : {precision*100:.2f}%')\nprint(f'Recall    : {recall*100:.2f}%')\nprint(f'F1-Score  : {f1*100:.2f}%')\nprint(f'AUC       : {auc*100:.2f}%')\n\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred, target_names=CLASS_NAMES))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:47:31.969160Z","iopub.execute_input":"2026-05-06T21:47:31.969447Z","iopub.status.idle":"2026-05-06T21:47:32.305386Z","shell.execute_reply.started":"2026-05-06T21:47:31.969422Z","shell.execute_reply":"2026-05-06T21:47:32.304498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 8 · Confusion Matrix\n# =========================\ncm = confusion_matrix(y_test, y_pred)\n\nplt.figure(figsize=(6,5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=CLASS_NAMES,\n            yticklabels=CLASS_NAMES)\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:47:32.307050Z","iopub.execute_input":"2026-05-06T21:47:32.307368Z","iopub.status.idle":"2026-05-06T21:47:32.478456Z","shell.execute_reply.started":"2026-05-06T21:47:32.307344Z","shell.execute_reply":"2026-05-06T21:47:32.477664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 9 · Summary Chart\n# =========================\nmetrics = {\n    'Accuracy': accuracy,\n    'Precision': precision,\n    'Recall': recall,\n    'F1': f1,\n    'AUC': auc\n}\n\nplt.bar(metrics.keys(), [v*100 for v in metrics.values()])\nplt.title('RSNA EfficientNet + RF')\nplt.ylabel('Score (%)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T21:47:32.479449Z","iopub.execute_input":"2026-05-06T21:47:32.479793Z","iopub.status.idle":"2026-05-06T21:47:32.589031Z","shell.execute_reply.started":"2026-05-06T21:47:32.479728Z","shell.execute_reply":"2026-05-06T21:47:32.588223Z"}},"outputs":[],"execution_count":null}]}