{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.listdir(\"/kaggle/input/competitions\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T19:55:15.937115Z","iopub.execute_input":"2026-02-19T19:55:15.937737Z","iopub.status.idle":"2026-02-19T19:55:15.941974Z","shell.execute_reply.started":"2026-02-19T19:55:15.937692Z","shell.execute_reply":"2026-02-19T19:55:15.941298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir(\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T19:55:26.687186Z","iopub.execute_input":"2026-02-19T19:55:26.687486Z","iopub.status.idle":"2026-02-19T19:55:26.691869Z","shell.execute_reply.started":"2026-02-19T19:55:26.687462Z","shell.execute_reply":"2026-02-19T19:55:26.691227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nDATA_DIR = \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge\"\nprint(os.listdir(DATA_DIR))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T19:58:27.771080Z","iopub.execute_input":"2026-02-19T19:58:27.771785Z","iopub.status.idle":"2026-02-19T19:58:27.777218Z","shell.execute_reply.started":"2026-02-19T19:58:27.771756Z","shell.execute_reply":"2026-02-19T19:58:27.776595Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain_df = pd.read_csv(DATA_DIR + \"/stage_2_train_labels.csv\")\nprint(train_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:00:37.519815Z","iopub.execute_input":"2026-02-19T20:00:37.520548Z","iopub.status.idle":"2026-02-19T20:00:37.592078Z","shell.execute_reply.started":"2026-02-19T20:00:37.520518Z","shell.execute_reply":"2026-02-19T20:00:37.591306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = train_df.groupby(\"patientId\")[\"Target\"].max().reset_index()\nlabels.rename(columns={\"patientId\": \"PatientID\"}, inplace=True)\nlabels.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:00:57.976876Z","iopub.execute_input":"2026-02-19T20:00:57.977441Z","iopub.status.idle":"2026-02-19T20:00:58.026591Z","shell.execute_reply.started":"2026-02-19T20:00:57.977413Z","shell.execute_reply":"2026-02-19T20:00:58.025971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels[\"path\"] = labels[\"PatientID\"].apply(\n    lambda x: f\"{DATA_DIR}/stage_2_train_images/{x}.dcm\"\n)\nlabels[\"label\"] = labels[\"Target\"].astype(int)\n\nlabels = labels[[\"path\", \"label\"]]\nprint(labels.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:01:07.839328Z","iopub.execute_input":"2026-02-19T20:01:07.839838Z","iopub.status.idle":"2026-02-19T20:01:07.857714Z","shell.execute_reply.started":"2026-02-19T20:01:07.839806Z","shell.execute_reply":"2026-02-19T20:01:07.856900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.path.exists(labels[\"path\"].iloc[0]))  # should print True\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:01:19.028901Z","iopub.execute_input":"2026-02-19T20:01:19.029647Z","iopub.status.idle":"2026-02-19T20:01:19.040903Z","shell.execute_reply.started":"2026-02-19T20:01:19.029618Z","shell.execute_reply":"2026-02-19T20:01:19.040304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pydicom\nimport pydicom\nimport numpy as np\nimport cv2\n\nIMG_SIZE = 300  # EfficientNetB3 input size\n\ndef load_dicom_image(path):\n    dcm = pydicom.dcmread(path)\n    img = dcm.pixel_array.astype(np.float32)\n\n    # Normalize\n    img = (img - img.min()) / (img.max() - img.min() + 1e-7)\n\n    # Resize\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n\n    # Convert grayscale -> 3 channels\n    img = np.stack([img, img, img], axis=-1)\n    return img\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:02:20.113620Z","iopub.execute_input":"2026-02-19T20:02:20.114268Z","iopub.status.idle":"2026-02-19T20:02:24.201734Z","shell.execute_reply.started":"2026-02-19T20:02:20.114237Z","shell.execute_reply":"2026-02-19T20:02:24.201043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_paths, val_paths, train_labels, val_labels = train_test_split(\n    labels[\"path\"].values,\n    labels[\"label\"].values,\n    test_size=0.2,\n    random_state=42,\n    stratify=labels[\"label\"].values\n)\n\nprint(len(train_paths), len(val_paths))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:02:35.424426Z","iopub.execute_input":"2026-02-19T20:02:35.425010Z","iopub.status.idle":"2026-02-19T20:02:35.557177Z","shell.execute_reply.started":"2026-02-19T20:02:35.424975Z","shell.execute_reply":"2026-02-19T20:02:35.556383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nBATCH_SIZE = 16\n\ndef data_generator(paths, labels, batch_size=BATCH_SIZE):\n    while True:\n        idxs = np.random.permutation(len(paths))\n        for i in range(0, len(paths), batch_size):\n            batch_idxs = idxs[i:i+batch_size]\n            batch_paths = paths[batch_idxs]\n            batch_labels = labels[batch_idxs]\n\n            images = [load_dicom_image(p) for p in batch_paths]\n            images = np.array(images)\n            batch_labels = np.array(batch_labels)\n\n            yield images, batch_labels\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:03:01.519157Z","iopub.execute_input":"2026-02-19T20:03:01.519809Z","iopub.status.idle":"2026-02-19T20:03:01.525474Z","shell.execute_reply.started":"2026-02-19T20:03:01.519777Z","shell.execute_reply":"2026-02-19T20:03:01.524582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB3\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Model\n\nbase_model = EfficientNetB3(\n    weights=\"imagenet\",\n    include_top=False,\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\n\n# Freeze backbone first (transfer learning)\nbase_model.trainable = False\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.3)(x)\noutput = Dense(1, activation=\"sigmoid\")(x)\n\nmodel = Model(inputs=base_model.input, outputs=output)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-4),\n    loss=\"binary_crossentropy\",\n    metrics=[\"accuracy\", tf.keras.metrics.AUC(name=\"auc\")]\n)\n\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:04:32.315482Z","iopub.execute_input":"2026-02-19T20:04:32.315908Z","iopub.status.idle":"2026-02-19T20:04:41.560921Z","shell.execute_reply.started":"2026-02-19T20:04:32.315880Z","shell.execute_reply":"2026-02-19T20:04:41.560189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 5  # start small\n\nhistory = model.fit(\n    data_generator(train_paths, train_labels),\n    steps_per_epoch=len(train_paths)//BATCH_SIZE,\n    validation_data=data_generator(val_paths, val_labels),\n    validation_steps=len(val_paths)//BATCH_SIZE,\n    epochs=EPOCHS\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:05:09.940414Z","iopub.execute_input":"2026-02-19T20:05:09.940763Z","iopub.status.idle":"2026-02-19T20:31:27.379590Z","shell.execute_reply.started":"2026-02-19T20:05:09.940736Z","shell.execute_reply":"2026-02-19T20:31:27.378745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"/kaggle/working/rsna_efficientnetb3.h5\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T20:34:06.050599Z","iopub.execute_input":"2026-02-19T20:34:06.050932Z","iopub.status.idle":"2026-02-19T20:34:06.701464Z","shell.execute_reply.started":"2026-02-19T20:34:06.050907Z","shell.execute_reply":"2026-02-19T20:34:06.700827Z"}},"outputs":[],"execution_count":null}]}