{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20604,"databundleVersionId":1357052,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:20:42.509753Z","iopub.execute_input":"2025-11-11T15:20:42.51038Z","iopub.status.idle":"2025-11-11T15:22:13.619988Z","shell.execute_reply.started":"2025-11-11T15:20:42.510356Z","shell.execute_reply":"2025-11-11T15:22:13.619029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 0 — quick checks & config\nimport os, sys\nprint(\"Python:\", sys.version.split()[0])\n!nvidia-smi -L || true\n\nROOT = \"/kaggle/input/osic-pulmonary-fibrosis-progression\"\nTRAIN_CSV = os.path.join(ROOT, \"train.csv\")\nTEST_CSV  = os.path.join(ROOT, \"test.csv\")\nBASE_DICOM = os.path.join(ROOT, \"train\")   # folder with patient subfolders\n\n# working cache\nIMG_FEAT_DIR = \"/kaggle/working/img_feats\"\nos.makedirs(IMG_FEAT_DIR, exist_ok=True)\n\n# hyperparams you can tune\nIMG_SIZE = 224\nNUM_SLICES = 5   # slices sampled per patient (reduce to 3 to speed up)\nBATCH_SIZE = 8\nEPOCHS = 8\n\nprint(\"ROOT:\", ROOT)\nprint(\"Train CSV exists:\", os.path.exists(TRAIN_CSV))\nprint(\"DICOM base exists:\", os.path.exists(BASE_DICOM))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:13.621585Z","iopub.execute_input":"2025-11-11T15:22:13.622167Z","iopub.status.idle":"2025-11-11T15:22:13.786947Z","shell.execute_reply.started":"2025-11-11T15:22:13.622143Z","shell.execute_reply":"2025-11-11T15:22:13.785916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 1 — install extras if missing (Kaggle typically already has these)\n!pip install -q pydicom opencv-python-headless efficientnet\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:13.788125Z","iopub.execute_input":"2025-11-11T15:22:13.788534Z","iopub.status.idle":"2025-11-11T15:22:23.614953Z","shell.execute_reply.started":"2025-11-11T15:22:13.788503Z","shell.execute_reply":"2025-11-11T15:22:23.614047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 2 — imports\nimport numpy as np\nimport pandas as pd\nimport pydicom, cv2, glob\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Model, Input\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nimport matplotlib.pyplot as plt\n\nprint(\"tf:\", tf.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:23.617172Z","iopub.execute_input":"2025-11-11T15:22:23.617461Z","iopub.status.idle":"2025-11-11T15:22:42.852387Z","shell.execute_reply.started":"2025-11-11T15:22:23.617432Z","shell.execute_reply":"2025-11-11T15:22:42.851577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 3 — load csv and quick EDA\ntrain_df = pd.read_csv(TRAIN_CSV)\ntest_df  = pd.read_csv(TEST_CSV)\nprint(\"train shape:\", train_df.shape)\ndisplay(train_df.head())\nprint(\"patients:\", train_df['Patient'].nunique())\ntrain_df.groupby('Patient').size().describe()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:42.85319Z","iopub.execute_input":"2025-11-11T15:22:42.853772Z","iopub.status.idle":"2025-11-11T15:22:42.906297Z","shell.execute_reply.started":"2025-11-11T15:22:42.853748Z","shell.execute_reply":"2025-11-11T15:22:42.905604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 4 — DICOM utilities\ndef load_scan(patient_dir):\n    files = [f for f in glob.glob(patient_dir + \"/**/*.dcm\", recursive=True)]\n    if not files:\n        return []\n    slices = [pydicom.dcmread(f) for f in files]\n    slices = sorted(slices, key=lambda s: getattr(s,'InstanceNumber',0))\n    return slices\n\ndef dicom_to_hu(scans):\n    imgs = np.stack([s.pixel_array for s in scans]).astype(np.int16)\n    for i, s in enumerate(scans):\n        intercept = getattr(s,'RescaleIntercept',0.0)\n        slope = getattr(s,'RescaleSlope',1.0)\n        if slope != 1:\n            imgs[i] = slope * imgs[i].astype(np.float64)\n            imgs[i] = imgs[i].astype(np.int16)\n        imgs[i] = imgs[i] + np.int16(intercept)\n    return imgs\n\ndef window_image(img, center=-600, width=1500):\n    minv = center - width//2\n    maxv = center + width//2\n    img = np.clip(img, minv, maxv)\n    img = (img - minv) / (maxv - minv)\n    img = (img * 255).astype(np.uint8)\n    return img\n\ndef preprocess_slice(img, out_size=IMG_SIZE):\n    img = cv2.resize(img, (out_size,out_size), interpolation=cv2.INTER_AREA)\n    img = np.stack([img, img, img], -1).astype(np.float32) / 255.0\n    return img\n\ndef select_slices(patient_dir, num_slices=NUM_SLICES):\n    scans = load_scan(patient_dir)\n    if len(scans) == 0:\n        return []\n    hu = dicom_to_hu(scans)\n    z = hu.shape[0]\n    if z <= num_slices:\n        idxs = list(range(z))\n    else:\n        idxs = np.linspace(0, z-1, num_slices, dtype=int).tolist()\n    return [hu[i] for i in idxs]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:42.907222Z","iopub.execute_input":"2025-11-11T15:22:42.907601Z","iopub.status.idle":"2025-11-11T15:22:42.91876Z","shell.execute_reply.started":"2025-11-11T15:22:42.90757Z","shell.execute_reply":"2025-11-11T15:22:42.91788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 5 — build EfficientNet base for feature extraction\nbase_cnn = EfficientNetB0(weights='imagenet', include_top=False, pooling='avg', input_shape=(IMG_SIZE,IMG_SIZE,3))\nimg_feat_dim = base_cnn.output_shape[1]\nprint(\"Image feature dim:\", img_feat_dim)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:42.919893Z","iopub.execute_input":"2025-11-11T15:22:42.920733Z","iopub.status.idle":"2025-11-11T15:22:47.271887Z","shell.execute_reply.started":"2025-11-11T15:22:42.920689Z","shell.execute_reply":"2025-11-11T15:22:47.271045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install decompressors for JPEG Lossless DICOMs\n!pip install -q pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg python-gdcm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:47.272758Z","iopub.execute_input":"2025-11-11T15:22:47.273059Z","iopub.status.idle":"2025-11-11T15:22:53.255188Z","shell.execute_reply.started":"2025-11-11T15:22:47.273039Z","shell.execute_reply":"2025-11-11T15:22:53.254271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom, glob, os\n# change to a known patient folder (use one present in /kaggle/input)\nsample = glob.glob(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train/*/*\", recursive=True)\nsample = sample[:10]\nprint(\"sample files:\", sample[:3])\nfor f in sample:\n    try:\n        ds = pydicom.dcmread(f)\n        arr = ds.pixel_array  # triggers decompression\n        print(\"OK:\", os.path.basename(f), \"shape:\", arr.shape)\n        break\n    except Exception as e:\n        print(\"Failed to read\", os.path.basename(f), \"->\", type(e).__name__, e)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:53.256396Z","iopub.execute_input":"2025-11-11T15:22:53.256689Z","iopub.status.idle":"2025-11-11T15:22:53.529528Z","shell.execute_reply.started":"2025-11-11T15:22:53.256663Z","shell.execute_reply":"2025-11-11T15:22:53.528807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pydicom, glob, os, cv2\nfrom tqdm import tqdm\n\ndef safe_dcmread(path):\n    try:\n        return pydicom.dcmread(path)\n    except Exception as e:\n        return None\n\ndef load_scan(patient_dir):\n    files = [f for f in glob.glob(os.path.join(patient_dir, \"**\", \"*.dcm\"), recursive=True)]\n    if not files:\n        return []\n    slices = []\n    for f in files:\n        ds = safe_dcmread(f)\n        if ds is None:\n            continue\n        slices.append((ds, f))\n    # sort by InstanceNumber when available; fallback to filename\n    slices = sorted(slices, key=lambda x: getattr(x[0], 'InstanceNumber', 0))\n    return [s[0] for s in slices]\n\ndef dicom_to_hu(scans):\n    imgs = []\n    for s in scans:\n        try:\n            arr = s.pixel_array  # may raise if decompression missing\n        except Exception as e:\n            # skip unreadable slice\n            continue\n        imgs.append(arr.astype(np.int16))\n    if len(imgs)==0:\n        return np.array([])\n    imgs = np.stack(imgs, axis=0)\n    for i, s in enumerate(scans[:imgs.shape[0]]):\n        intercept = getattr(s, 'RescaleIntercept', 0.0)\n        slope = getattr(s, 'RescaleSlope', 1.0)\n        if slope != 1:\n            imgs[i] = slope * imgs[i].astype(np.float64)\n            imgs[i] = imgs[i].astype(np.int16)\n        imgs[i] = imgs[i] + np.int16(intercept)\n    return imgs\n\ndef window_image(img, center=-600, width=1500):\n    minv = center - width//2\n    maxv = center + width//2\n    img = np.clip(img, minv, maxv)\n    img = (img - minv) / (maxv - minv)\n    img = (img * 255).astype(np.uint8)\n    return img\n\ndef preprocess_slice(img, out_size=224):\n    img = cv2.resize(img, (out_size, out_size), interpolation=cv2.INTER_AREA)\n    img = np.stack([img, img, img], -1).astype(np.float32) / 255.0\n    return img\n\ndef select_slices(patient_dir, num_slices=5):\n    scans = load_scan(patient_dir)\n    if len(scans) == 0:\n        return []\n    hu = dicom_to_hu(scans)\n    if hu.size == 0:\n        return []\n    z = hu.shape[0]\n    if z <= num_slices:\n        idxs = list(range(z))\n    else:\n        idxs = np.linspace(0, z-1, num_slices, dtype=int).tolist()\n    return [hu[i] for i in idxs]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:53.53234Z","iopub.execute_input":"2025-11-11T15:22:53.532592Z","iopub.status.idle":"2025-11-11T15:22:53.544619Z","shell.execute_reply.started":"2025-11-11T15:22:53.532539Z","shell.execute_reply":"2025-11-11T15:22:53.543785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 6 — caching image features (with internal try/except)\ndef find_patient_folder(pid):\n    direct = os.path.join(BASE_DICOM, pid)\n    if os.path.isdir(direct):\n        return direct\n    res = glob.glob(os.path.join(BASE_DICOM, \"**\", pid), recursive=True)\n    return res[0] if res else None\n\ndef extract_patient_feature(pid):\n    out_fp = f\"{IMG_FEAT_DIR}/{pid}_feat.npy\"\n    if os.path.exists(out_fp):\n        return np.load(out_fp)\n    try:\n        pdir = find_patient_folder(pid)\n        if pdir is None:\n            raise ValueError(\"No folder found for patient\")\n        slices = select_slices(pdir)\n        if len(slices) == 0:\n            raise ValueError(\"No readable slices\")\n        feats = []\n        for s in slices:\n            img = window_image(s)\n            img = preprocess_slice(img)\n            emb = base_cnn.predict(np.expand_dims(img,0), verbose=0)\n            feats.append(emb[0])\n        feats = np.stack(feats,0)\n        avg = feats.mean(axis=0)\n    except Exception as e:\n        print(f\"Skipping {pid}: {e}\")\n        avg = np.zeros(img_feat_dim, dtype=np.float32)\n    np.save(out_fp, avg)\n    return avg\n\nunique_patients = train_df['Patient'].unique().tolist()\nfor pid in tqdm(unique_patients):\n    _ = extract_patient_feature(pid)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:22:53.545472Z","iopub.execute_input":"2025-11-11T15:22:53.545958Z","iopub.status.idle":"2025-11-11T15:33:11.308386Z","shell.execute_reply.started":"2025-11-11T15:22:53.545939Z","shell.execute_reply":"2025-11-11T15:33:11.307576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lh $IMG_FEAT_DIR | head\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:11.309293Z","iopub.execute_input":"2025-11-11T15:33:11.309598Z","iopub.status.idle":"2025-11-11T15:33:11.488451Z","shell.execute_reply.started":"2025-11-11T15:33:11.309569Z","shell.execute_reply":"2025-11-11T15:33:11.487596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 7 — attach features to rows\ndef add_img_feats_to_df(df):\n    feats = np.stack([np.load(os.path.join(IMG_FEAT_DIR, f\"{pid}_feat.npy\")) for pid in df['Patient'].values])\n    cols = [f\"img_{i}\" for i in range(feats.shape[1])]\n    feats_df = pd.DataFrame(feats, columns=cols)\n    res = pd.concat([df.reset_index(drop=True), feats_df.reset_index(drop=True)], axis=1)\n    return res\n\ntrain = add_img_feats_to_df(train_df)\ntest = add_img_feats_to_df(test_df)\nprint(\"train with img shape:\", train.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:11.489773Z","iopub.execute_input":"2025-11-11T15:33:11.490185Z","iopub.status.idle":"2025-11-11T15:33:11.728072Z","shell.execute_reply.started":"2025-11-11T15:33:11.490156Z","shell.execute_reply":"2025-11-11T15:33:11.727136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 8 — tabular processing\ntrain = train.copy()\ntest = test.copy()\ntrain['Sex'] = train['Sex'].map({'Male':1,'Female':0})\ntest['Sex'] = test['Sex'].map({'Male':1,'Female':0})\n\n# create baseline FVC per patient if needed\nbase_fvc = train.groupby('Patient').first().reset_index()[['Patient','FVC']].rename(columns={'FVC':'Baseline_FVC'})\ntrain = train.merge(base_fvc, on='Patient', how='left')\ntest = test.merge(base_fvc, on='Patient', how='left')\n\nTAB_FEATURES = ['Age','Sex','Weeks','Baseline_FVC','Percent']\nIMG_FEATURES = [c for c in train.columns if c.startswith('img_')]\n\nscaler = StandardScaler()\ntrain[TAB_FEATURES] = scaler.fit_transform(train[TAB_FEATURES].fillna(-999))\ntest[TAB_FEATURES]  = scaler.transform(test[TAB_FEATURES].fillna(-999))\n\nX_img = train[IMG_FEATURES].values.astype(np.float32)\nX_tab = train[TAB_FEATURES].values.astype(np.float32)\ny = train['FVC'].values.astype(np.float32)\nprint(\"X_img shape:\", X_img.shape, \"X_tab shape:\", X_tab.shape, \"y shape:\", y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:11.729183Z","iopub.execute_input":"2025-11-11T15:33:11.729522Z","iopub.status.idle":"2025-11-11T15:33:11.798118Z","shell.execute_reply.started":"2025-11-11T15:33:11.729494Z","shell.execute_reply":"2025-11-11T15:33:11.79729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 9 — build model\ndef build_fusion_model(img_dim, tab_dim):\n    img_in = Input(shape=(img_dim,), name='img_in')\n    x1 = layers.Dense(256, activation='relu')(img_in)\n    x1 = layers.Dropout(0.3)(x1)\n\n    tab_in = Input(shape=(tab_dim,), name='tab_in')\n    x2 = layers.Dense(64, activation='relu')(tab_in)\n    x2 = layers.Dropout(0.2)(x2)\n\n    x = layers.Concatenate()([x1, x2])\n    x = layers.Dense(128, activation='relu')(x)\n    x = layers.Dropout(0.25)(x)\n    x = layers.Dense(64, activation='relu')(x)\n\n    fvc_out = layers.Dense(1, name='fvc')(x)\n    sigma_out = layers.Dense(1, activation='softplus', name='sigma')(x)\n\n    model = Model(inputs=[img_in, tab_in], outputs=[fvc_out, sigma_out])\n    return model\n\nfusion_model = build_fusion_model(len(IMG_FEATURES), len(TAB_FEATURES))\nfusion_model.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:11.799163Z","iopub.execute_input":"2025-11-11T15:33:11.799776Z","iopub.status.idle":"2025-11-11T15:33:11.914711Z","shell.execute_reply.started":"2025-11-11T15:33:11.799751Z","shell.execute_reply":"2025-11-11T15:33:11.913969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 10 — subclassed model with Laplace NLL\nimport tensorflow.keras.backend as K\n\nclass LaplaceModel(Model):\n    def train_step(self, data):\n        (img_b, tab_b), y_b = data\n        with tf.GradientTape() as tape:\n            y_pred, sigma = self([img_b, tab_b], training=True)\n            y_pred = tf.squeeze(y_pred, -1)\n            sigma = tf.maximum(sigma, 1e-3)\n            diff = tf.abs(y_b - y_pred)\n            loss = tf.reduce_mean(tf.sqrt(2.0) * diff / sigma + tf.math.log(tf.sqrt(2.0) * sigma))\n        grads = tape.gradient(loss, self.trainable_variables)\n        self.optimizer.apply_gradients(zip(grads, self.trainable_variables))\n        mae = tf.reduce_mean(tf.abs(y_b - y_pred))\n        return {\"loss\": loss, \"mae\": mae}\n\n    def test_step(self, data):\n        (img_b, tab_b), y_b = data\n        y_pred, sigma = self([img_b, tab_b], training=False)\n        y_pred = tf.squeeze(y_pred, -1)\n        sigma = tf.maximum(sigma, 1e-3)\n        diff = tf.abs(y_b - y_pred)\n        loss = tf.reduce_mean(tf.sqrt(2.0) * diff / sigma + tf.math.log(tf.sqrt(2.0) * sigma))\n        mae = tf.reduce_mean(tf.abs(y_b - y_pred))\n        return {\"loss\": loss, \"mae\": mae}\n\nlaplace_model = LaplaceModel(inputs=fusion_model.inputs, outputs=fusion_model.outputs)\nlaplace_model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:11.915717Z","iopub.execute_input":"2025-11-11T15:33:11.916136Z","iopub.status.idle":"2025-11-11T15:33:11.940643Z","shell.execute_reply.started":"2025-11-11T15:33:11.916106Z","shell.execute_reply":"2025-11-11T15:33:11.939697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11 — one GroupKFold split (patient-wise) and training\ngkf = GroupKFold(n_splits=5)\ntrain_idx, val_idx = next(gkf.split(train, train['FVC'], groups=train['Patient']))\n\nXimg_tr, Ximg_va = X_img[train_idx], X_img[val_idx]\nXtab_tr, Xtab_va = X_tab[train_idx], X_tab[val_idx]\ny_tr, y_va = y[train_idx], y[val_idx]\n\ntrain_ds = tf.data.Dataset.from_tensor_slices(((Ximg_tr, Xtab_tr), y_tr)).shuffle(2048).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nval_ds   = tf.data.Dataset.from_tensor_slices(((Ximg_va, Xtab_va), y_va)).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\nhistory = laplace_model.fit(train_ds, validation_data=val_ds, epochs=EPOCHS)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:11.941597Z","iopub.execute_input":"2025-11-11T15:33:11.94191Z","iopub.status.idle":"2025-11-11T15:33:21.773136Z","shell.execute_reply.started":"2025-11-11T15:33:11.941884Z","shell.execute_reply":"2025-11-11T15:33:21.772224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 12 — validate\npreds_fvc, preds_sigma = laplace_model.predict([Ximg_va, Xtab_va], batch_size=BATCH_SIZE)\npreds_fvc = preds_fvc.flatten()\npreds_sigma = np.maximum(preds_sigma.flatten(), 1e-3)\n\nmae = mean_absolute_error(y_va, preds_fvc)\nrmse = mean_squared_error(y_va, preds_fvc, squared=False)\nprint(\"VAL MAE:\", mae, \"RMSE:\", rmse)\n\n# show some examples\nfor i in range(5):\n    print(\"True:\", y_va[i], \"Pred:\", float(preds_fvc[i]), \"Sigma:\", float(preds_sigma[i]))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:21.774241Z","iopub.execute_input":"2025-11-11T15:33:21.775044Z","iopub.status.idle":"2025-11-11T15:33:22.484993Z","shell.execute_reply.started":"2025-11-11T15:33:21.775018Z","shell.execute_reply":"2025-11-11T15:33:22.483921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 13 — prepare test arrays and predict\nX_img_test = test[IMG_FEATURES].values.astype(np.float32)\nX_tab_test = test[TAB_FEATURES].values.astype(np.float32)\n\npreds_fvc_test, preds_sigma_test = laplace_model.predict([X_img_test, X_tab_test], batch_size=BATCH_SIZE)\npreds_fvc_test = preds_fvc_test.flatten()\npreds_sigma_test = np.clip(preds_sigma_test.flatten(), 70, None)\n\n# Build Patient_Week key used in sample submission\ntest['Patient_Week'] = test['Patient'] + '_' + test['Weeks'].astype(int).astype(str)\nsub = pd.read_csv(os.path.join(ROOT, \"sample_submission.csv\"))\npred_map = pd.DataFrame({'Patient_Week': test['Patient_Week'], 'FVC': preds_fvc_test, 'Confidence': preds_sigma_test})\nif 'Patient_Week' in sub.columns:\n    sub = sub.drop(columns=[c for c in ['FVC','Confidence'] if c in sub.columns], errors='ignore')\n    sub = sub.merge(pred_map, on='Patient_Week', how='left')\nelse:\n    sub['FVC'] = preds_fvc_test\n    sub['Confidence'] = preds_sigma_test\n\nsub['FVC'] = sub['FVC'].fillna(sub['FVC'].mean())\nsub['Confidence'] = sub['Confidence'].fillna(sub['Confidence'].mean())\nsub.to_csv(\"/kaggle/working/submission_with_images.csv\", index=False)\nprint(\"Saved /kaggle/working/submission_with_images.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:22.486206Z","iopub.execute_input":"2025-11-11T15:33:22.487258Z","iopub.status.idle":"2025-11-11T15:33:22.624439Z","shell.execute_reply.started":"2025-11-11T15:33:22.487222Z","shell.execute_reply":"2025-11-11T15:33:22.623475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 14 — demo: choose a patient and plot predicted curve over weeks\npid = train['Patient'].unique()[0]\nrows = train[train['Patient']==pid]\nXimg_p = rows[IMG_FEATURES].values.astype(np.float32)\nXtab_p = rows[TAB_FEATURES].values.astype(np.float32)\npf, ps = laplace_model.predict([Ximg_p, Xtab_p])\nweeks = rows['Weeks'].values\nplt.figure(figsize=(6,4))\nplt.errorbar(weeks, pf.flatten(), yerr=ps.flatten(), fmt='-o', label='Pred FVC +/- sigma')\nplt.scatter(weeks, rows['FVC'].values, color='k', alpha=0.6, label='True FVC')\nplt.xlabel(\"Weeks\"); plt.ylabel(\"FVC\"); plt.legend(); plt.show()\n\nresult_table = pd.DataFrame({\n    \"Week\": weeks.astype(int),\n    \"True_FVC (mL)\": np.round(rows['FVC'].values, 1),\n    \"Predicted_FVC (mL)\": np.round(pf.flatten(), 1),\n    \"Sigma (Uncertainty)\": np.round(ps.flatten(), 1)\n}).sort_values(\"Week\").reset_index(drop=True)\n\nprint(f\"\\nPatient ID: {pid}\")\ndisplay(result_table.head(20))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T17:03:38.728809Z","iopub.execute_input":"2025-11-11T17:03:38.72962Z","iopub.status.idle":"2025-11-11T17:03:39.016814Z","shell.execute_reply.started":"2025-11-11T17:03:38.72959Z","shell.execute_reply":"2025-11-11T17:03:39.015852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fix: show ORIGINAL (raw) Weeks alongside model predictions\npid = train['Patient'].unique()[0]   # same patient used for plotting\n# ensure pf and ps exist (predictions from the model for this patient's rows)\n# rows = train[train['Patient']==pid]  # you already have this from plotting cell\n\n# Load raw train (unscaled) if not already available\nraw_train = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\n# Get raw rows for the patient and sort by raw Weeks\nraw_rows = raw_train[raw_train['Patient'] == pid].sort_values('Weeks').reset_index(drop=True)\n\n# Ensure prediction arrays are the same length & aligned — we assume your 'rows' used for prediction\n# were ordered by Weeks as well. If not, sort 'rows' the same way as raw_rows:\nrows_sorted = rows.copy().reset_index(drop=True)  # 'rows' came from the scaled train df\n# If lengths mismatch, print diagnostic\nif len(rows_sorted) != len(raw_rows):\n    print(\"WARNING: row counts differ (scaled vs raw). Aligning by nearest week values.\")\n    # try simple alignment by matching unique Weeks if possible:\n    raw_weeks = raw_rows['Weeks'].values\n    scaled_weeks = rows_sorted['Weeks'].values\n    print(\"raw weeks sample:\", raw_weeks[:10])\n    print(\"scaled weeks sample:\", scaled_weeks[:10])\n\n# Build results DataFrame using raw weeks and true FVC\nresults_table = pd.DataFrame({\n    \"Week (raw)\": raw_rows['Weeks'].values,\n    \"True_FVC (mL)\": raw_rows['FVC'].values,\n    \"Predicted_FVC (mL)\": np.round(pf.flatten(), 1),\n    \"Sigma (Uncertainty)\": np.round(ps.flatten(), 1)\n})\n\nprint(f\"Patient ID: {pid}\")\ndisplay(results_table)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T17:09:15.592997Z","iopub.execute_input":"2025-11-11T17:09:15.594123Z","iopub.status.idle":"2025-11-11T17:09:15.625885Z","shell.execute_reply.started":"2025-11-11T17:09:15.594092Z","shell.execute_reply":"2025-11-11T17:09:15.624956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pick a patient ID from your training data\npatient_id = train['Patient'].iloc[0]  # or replace with any ID string\n\n# Filter the rows for that patient\npatient_rows = train[train['Patient'] == patient_id]\n\n# Prepare arrays\nXimg_p = patient_rows[[c for c in train.columns if c.startswith('img_')]].values.astype(np.float32)\nXtab_p = patient_rows[['Age','Sex','Weeks','Baseline_FVC','Percent']].values.astype(np.float32)\n\n# Predict FVC and sigma for all weeks of this patient\npred_fvc, pred_sigma = laplace_model.predict([Ximg_p, Xtab_p], verbose=0)\npred_fvc = pred_fvc.flatten()\npred_sigma = np.clip(pred_sigma.flatten(), 70, None)\n\n# Combine into a DataFrame\npred_df = pd.DataFrame({\n    'Weeks': patient_rows['Weeks'].values,\n    'True_FVC': patient_rows['FVC'].values,\n    'Predicted_FVC': pred_fvc,\n    'Sigma': pred_sigma\n}).sort_values('Weeks')\n\ndisplay(pred_df.head(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:23.386171Z","iopub.execute_input":"2025-11-11T15:33:23.386476Z","iopub.status.idle":"2025-11-11T15:33:23.495947Z","shell.execute_reply.started":"2025-11-11T15:33:23.386456Z","shell.execute_reply":"2025-11-11T15:33:23.495187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,5))\nplt.errorbar(pred_df['Weeks'], pred_df['Predicted_FVC'],\n             yerr=pred_df['Sigma'], fmt='-o', label='Predicted FVC ± σ', alpha=0.7)\nplt.scatter(pred_df['Weeks'], pred_df['True_FVC'], color='black', label='True FVC', zorder=5)\nplt.xlabel(\"Weeks\")\nplt.ylabel(\"FVC (mL)\")\nplt.title(f\"Patient: {patient_id}\")\nplt.legend()\nplt.grid(alpha=0.3)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:23.496825Z","iopub.execute_input":"2025-11-11T15:33:23.497129Z","iopub.status.idle":"2025-11-11T15:33:23.721894Z","shell.execute_reply.started":"2025-11-11T15:33:23.497103Z","shell.execute_reply":"2025-11-11T15:33:23.720976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ FIXED LIVE PREDICTION CELL — refits scaler from RAW data\n\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np, pandas as pd, os\n\n# ---- 1️⃣ Reload the ORIGINAL training CSV (raw numbers, not scaled) ----\nraw_train = pd.read_csv(\"/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv\")\n\nTAB_FEATURES = ['Age','Sex','Weeks','Baseline_FVC','Percent']\n\n# Encode Sex again on raw data\nraw_train['Sex'] = raw_train['Sex'].map({'Male':1,'Female':0})\n\n# Compute baseline FVC per patient\nbase_fvc = raw_train.groupby('Patient').first().reset_index()[['Patient','FVC']].rename(columns={'FVC':'Baseline_FVC'})\nraw_train = raw_train.merge(base_fvc, on='Patient', how='left')\n\n# ---- 2️⃣ Fit scaler on the *raw numeric columns* ----\nscaler = StandardScaler()\nscaler.fit(raw_train[TAB_FEATURES].fillna(-999))\nprint(\"✅ Scaler fitted on raw data.\")\nprint(\"Mean:\", np.round(scaler.mean_, 2))\nprint(\"Var:\", np.round(scaler.var_, 2))\n\n# ---- 3️⃣ Choose a patient and weeks to predict ----\npatient_id = train['Patient'].iloc[0]\nweeks_to_predict = [4, 8, 12]\n\n# Get baseline row for that patient (from raw data)\nbase_row = raw_train[raw_train['Patient'] == patient_id].sort_values('Weeks').iloc[0]\n\n# ---- 4️⃣ Build input table ----\npred_table = pd.DataFrame({\n    'Patient': [patient_id]*len(weeks_to_predict),\n    'Weeks': weeks_to_predict,\n    'Age': base_row['Age'],\n    'Sex': base_row['Sex'],\n    'SmokingStatus': base_row.get('SmokingStatus', 'Unknown'),\n    'Baseline_FVC': base_row['Baseline_FVC'],\n    'Percent': base_row['Percent']\n})\n\n# ---- 5️⃣ Apply scaling properly ----\nX_tab_live = scaler.transform(pred_table[TAB_FEATURES].fillna(-999)).astype(np.float32)\n\n# ---- 6️⃣ Load image features ----\nX_img_live = np.load(os.path.join(IMG_FEAT_DIR, f\"{patient_id}_feat.npy\")).reshape(1, -1)\nX_img_live = np.repeat(X_img_live, len(weeks_to_predict), axis=0)\n\n# ---- 7️⃣ Predict using the trained model ----\nfvc_pred, sigma_pred = laplace_model.predict([X_img_live, X_tab_live], verbose=0)\nfvc_pred = fvc_pred.flatten()\nsigma_pred = np.clip(sigma_pred.flatten(), 70, None)\n\n# ---- 8️⃣ Display table ----\nresults = pd.DataFrame({\n    \"Week\": weeks_to_predict,\n    \"Predicted_FVC (mL)\": np.round(fvc_pred, 1),\n    \"Sigma (Uncertainty)\": np.round(sigma_pred, 1)\n})\n\nprint(f\"\\nPatient ID: {patient_id}\")\ndisplay(results)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:23.722887Z","iopub.execute_input":"2025-11-11T15:33:23.72313Z","iopub.status.idle":"2025-11-11T15:33:24.152059Z","shell.execute_reply.started":"2025-11-11T15:33:23.723111Z","shell.execute_reply":"2025-11-11T15:33:24.151176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train_sample = y[:10]\nprint(\"Sample y (first 10):\", y_train_sample)\nprint(\"Mean:\", np.mean(y_train_sample), \"Std:\", np.std(y_train_sample))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T15:33:24.153101Z","iopub.execute_input":"2025-11-11T15:33:24.153869Z","iopub.status.idle":"2025-11-11T15:33:24.159256Z","shell.execute_reply.started":"2025-11-11T15:33:24.153846Z","shell.execute_reply":"2025-11-11T15:33:24.158385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CALIBRATE RAW MODEL OUTPUTS AND SHOW CALIBRATED TABLE FOR A PATIENT\nimport numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import mean_absolute_error\n\n# === 1) Recreate the same val split if needed ===\ntry:\n    Ximg_va, Xtab_va, y_va  # if these exist, keep them\nexcept NameError:\n    gkf = GroupKFold(n_splits=5)\n    train_idx, val_idx = next(gkf.split(train, train['FVC'], groups=train['Patient']))\n    # X_img and X_tab should be the arrays you created earlier in Cell 8\n    Ximg_va = X_img[val_idx]\n    Xtab_va = X_tab[val_idx]\n    y_va    = y[val_idx]\n\n# === 2) Get raw predictions on validation ===\npreds_val_fvc, preds_val_sigma = laplace_model.predict([Ximg_va, Xtab_va], batch_size=32, verbose=0)\npreds_val_fvc = preds_val_fvc.flatten()\npreds_val_sigma = preds_val_sigma.flatten()\n\nprint(\"Sample raw pred vs true (val):\")\nfor i in range(6):\n    print(f\" raw_pred={preds_val_fvc[i]:.4f}, true={y_va[i]:.1f}\")\n\n# === 3) Fit a linear calibration y_true = a * y_pred_raw + b ===\nlr = LinearRegression()\nlr.fit(preds_val_fvc.reshape(-1,1), y_va.reshape(-1,1))\na = float(lr.coef_[0][0])\nb = float(lr.intercept_[0])\nprint(f\"\\nCalibration: y_true ≈ a * y_pred_raw + b\")\nprint(f\" a = {a:.6f}, b = {b:.3f}\")\n\n# Report MAE before / after (on val)\npreds_val_cal = a * preds_val_fvc + b\nprint(\"Val MAE before calibration:\", mean_absolute_error(y_va, preds_val_fvc))\nprint(\"Val MAE after  calibration:\", mean_absolute_error(y_va, preds_val_cal))\n\n# === 4) Apply calibration to your patient predictions (the pred_df you showed) ===\n# If you still have pred_df (from previous cell) with columns ['Weeks','True_FVC','Predicted_FVC','Sigma'],\n# use it. Otherwise re-create predictions for the specific patient and weeks:\ntry:\n    pred_df  # exists\n    raw_pred = pred_df['Predicted_FVC'].values\n    raw_sigma = pred_df['Sigma'].values\nexcept NameError:\n    # recreate for chosen patient\n    patient_id = train['Patient'].iloc[0]\n    patient_rows = train[train['Patient'] == patient_id]\n    Ximg_p = patient_rows[[c for c in train.columns if c.startswith('img_')]].values.astype(np.float32)\n    Xtab_p = patient_rows[['Age','Sex','Weeks','Baseline_FVC','Percent']].values.astype(np.float32)\n    raw_pred, raw_sigma = laplace_model.predict([Ximg_p, Xtab_p], verbose=0)\n    raw_pred = raw_pred.flatten()\n    raw_sigma = np.clip(raw_sigma.flatten(), 0, None)\n\n# Calibrate\ncal_pred = a * raw_pred + b\ncal_sigma = np.clip(np.abs(a) * raw_sigma, 70, None)  # scale uncertainty and clip minimum\n\n# Build result table\nresults = pd.DataFrame({\n    \"Week\": pred_df['Weeks'].values if 'pred_df' in globals() else patient_rows['Weeks'].values,\n    \"True_FVC (mL)\": pred_df['True_FVC'].values if 'pred_df' in globals() else patient_rows['FVC'].values,\n    \"Predicted_FVC_raw\": np.round(raw_pred, 3),\n    \"Predicted_FVC_calibrated (mL)\": np.round(cal_pred, 1),\n    \"Sigma_raw\": np.round(raw_sigma, 3),\n    \"Sigma_calibrated\": np.round(cal_sigma, 1)\n})\nprint(\"\\nCalibrated predictions for patient:\")\ndisplay(results)\n\n# === 5) Quick check: if calibration factor is extreme, warn ===\nif abs(a) < 0.01 or abs(b) > 10000 or a > 100:\n    print(\"\\nWARNING: calibration factor looks extreme. Consider retraining the model on raw FVC targets or diagnosing training pipeline.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-11T17:02:15.001371Z","iopub.execute_input":"2025-11-11T17:02:15.002192Z","iopub.status.idle":"2025-11-11T17:02:15.923292Z","shell.execute_reply.started":"2025-11-11T17:02:15.002167Z","shell.execute_reply":"2025-11-11T17:02:15.922573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D, Concatenate\nfrom tensorflow.keras.applications import EfficientNetB3 \nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom typing import Tuple, Dict, List\nfrom glob import glob\nimport warnings\n\n# --- CONFIGURATION ---\nIMG_SIZE = 256\nTABULAR_FEATURES = ['Weeks', 'Age', 'FVC', 'Sex_male', 'SmokingStatus_Ex-smoker', 'SmokingStatus_Never Smoker']\nTABULAR_DIM = len(TABULAR_FEATURES)\n# --- FIXED: Changed extension to satisfy Keras 2.16+ naming convention ---\nMODEL_WEIGHTS_PATH = 'fvc_model_weights.weights.h5'\nEPOCHS = 5 \nBATCH_SIZE = 4\n\n# --- PATHS (Standard Kaggle OSIC paths) ---\nKAGGLE_DICOM_DIR = '../input/osic-pulmonary-fibrosis-progression/train'\nKAGGLE_CSV_PATH = '../input/osic-pulmonary-fibrosis-progression/train.csv'\n\n# ==============================================================================\n# 1. CUSTOM LOSS FUNCTION\n# ==============================================================================\n\ndef laplace_log_likelihood(y_true, y_pred):\n    \"\"\"\n    Custom Loss Function (Negative Log-Likelihood) for the two-head model.\n    The model predicts [mu (FVC), log(sigma)].\n    \"\"\"\n    mu = y_pred[:, 0]\n    log_sigma = y_pred[:, 1] \n    sigma = K.exp(log_sigma) \n    y_true_fvc = y_true[:, 0]\n    \n    # Laplace Negative Log-Likelihood formula\n    loss = (K.abs(y_true_fvc - mu) / sigma) + K.log(2 * sigma)\n    \n    return K.mean(loss)\n\n# ==============================================================================\n# 2. MODEL ARCHITECTURE DEFINITION\n# ==============================================================================\n\ndef build_model(input_shape=(IMG_SIZE, IMG_SIZE, 1), tabular_dim=TABULAR_DIM) -> Model:\n    \"\"\"\n    Defines the dual-input EfficientNet-B3 model with fusion layer.\n    \"\"\"\n    # --- CNN Branch (Image Input) ---\n    img_input = Input(shape=input_shape, name='image_input')\n    x = Concatenate()([img_input, img_input, img_input])\n\n    # Load ImageNet weights for transfer learning\n    cnn = EfficientNetB3(weights='imagenet', include_top=False, input_tensor=x)\n    \n    image_features = cnn.output\n    image_features = GlobalAveragePooling2D()(image_features)\n    image_features = Dense(64, activation='relu')(image_features)\n\n    # --- Tabular Branch (Metadata Input) ---\n    tabular_input = Input(shape=(tabular_dim,), name='tabular_input')\n    tabular_features = Dense(32, activation='relu')(tabular_input)\n\n    # --- Fusion and Output ---\n    fused = Concatenate()([image_features, tabular_features])\n    fused = Dense(64, activation='relu')(fused)\n    \n    # Output 1: mu (Predicted FVC)\n    mu_output = Dense(1, name='mu')(fused)\n    \n    # Output 2: log(sigma) (Predicted Uncertainty/Confidence)\n    sigma_output = Dense(1, name='log_sigma')(fused)\n    \n    combined_output = Concatenate(axis=-1, name='combined_output')([mu_output, sigma_output])\n    \n    model = Model(inputs=[img_input, tabular_input], outputs=combined_output)\n    \n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4), loss=laplace_log_likelihood)\n    \n    return model\n\n# ==============================================================================\n# 3. PREPROCESSING UTILITIES\n# ==============================================================================\n\ndef preprocess_dicom_path(dicom_path: str) -> np.ndarray:\n    \"\"\"\n    Reads a DICOM file from a path, applies Hounsfield Unit scaling and lung windowing, \n    and returns a normalized NumPy array ready for model input.\n    \"\"\"\n    try:\n        dcm_data = pydicom.dcmread(dicom_path)\n        img = dcm_data.pixel_array.astype(np.int16) # Convert to int16 for HU calculation\n        \n        # 1. Apply Rescaling and Intercept to get Hounsfield Units (HU)\n        if 'RescaleSlope' in dcm_data and 'RescaleIntercept' in dcm_data:\n            img = img * dcm_data.RescaleSlope + dcm_data.RescaleIntercept\n        \n        # 2. Lung Windowing and Normalization\n        # Standard range for lung tissue: -1000 HU to -400 HU\n        MIN_HU = -1000.0\n        MAX_HU = -400.0\n        \n        # Clip HU values to the lung window\n        img = np.clip(img, MIN_HU, MAX_HU)\n        \n        # Normalize the clipped image to the [0, 1] range\n        img = (img - MIN_HU) / (MAX_HU - MIN_HU)\n        \n        # 3. Resize and Finalize\n        img = (img * 255).astype(np.uint8) \n        img_resized = cv2.resize(img, (IMG_SIZE, IMG_SIZE), interpolation=cv2.INTER_LINEAR)\n        \n        # Final formatting: float32, normalized, and shaped (IMG_SIZE, IMG_SIZE, 1)\n        img_final = img_resized.astype(np.float32) / 255.0\n        img_final = np.expand_dims(img_final, axis=-1)\n        \n        return img_final\n\n    except Exception as e:\n        print(f\"DICOM processing error for path {dicom_path}: {e}\")\n        # Return a zero array upon failure\n        return np.zeros((IMG_SIZE, IMG_SIZE, 1), dtype=np.float32) \n\n\ndef prepare_tabular_data(df: pd.DataFrame) -> np.ndarray:\n    \"\"\"\n    Encodes categorical features and prepares the tabular data vector for training.\n    \"\"\"\n    df_copy = df.copy()\n    \n    # One-Hot Encoding (OHE)\n    # The Smoking Status and Sex labels must match the column names in TABULAR_FEATURES\n    df_copy = pd.get_dummies(df_copy, columns=['Sex', 'SmokingStatus'], prefix=['Sex', 'SmokingStatus'])\n\n    for col in TABULAR_FEATURES:\n        if col not in df_copy.columns:\n            df_copy[col] = 0\n\n    # Ensure final columns are in the correct order for the model\n    tabular_input_vector = df_copy[TABULAR_FEATURES].values.astype(np.float32)\n\n    return tabular_input_vector\n\n# ==============================================================================\n# 4. DATA LOADING AND ALIGNMENT\n# ==============================================================================\n\ndef load_data(dicom_dir: str, csv_path: str) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:\n    \"\"\"\n    Loads and aligns tabular data (from CSV) and image data (from DICOM directory).\n    \"\"\"\n    \n    if not os.path.exists(csv_path):\n        raise FileNotFoundError(f\"CSV file not found at: {csv_path}\")\n\n    full_df = pd.read_csv(csv_path)\n    train_df = full_df \n    \n    # 1. Prepare Tabular Input\n    X_tab_data = prepare_tabular_data(train_df)\n    NUM_SAMPLES = len(train_df)\n    \n    # 2. Prepare Target (y) Data\n    y_data = np.zeros((NUM_SAMPLES, 2), dtype=np.float32)\n    y_data[:, 0] = train_df['FVC'].values # True FVC is the target (mu)\n\n    # 3. Prepare Image Input (X_img_data)\n    \n    image_data_list = []\n    \n    # Create a mapping of Patient ID to the path of a single DICOM slice\n    patient_to_dicom = {}\n    \n    for patient_id in train_df['Patient'].unique():\n        dicom_folder = os.path.join(dicom_dir, patient_id)\n        \n        # Find all DICOM files in the patient's directory\n        dicom_files = sorted(glob(f\"{dicom_folder}/*.dcm\"))\n        \n        if dicom_files:\n            # Common strategy: Use the single center slice as a representative image\n            center_slice_path = dicom_files[len(dicom_files) // 2]\n            patient_to_dicom[patient_id] = center_slice_path\n        else:\n            print(f\"Warning: No DICOM files found for patient {patient_id}. Skipping.\")\n\n    # Now, process images based on the training data rows\n    processed_count = 0\n    \n    for index, row in train_df.iterrows():\n        patient_id = row['Patient']\n        if patient_id in patient_to_dicom:\n            dicom_path = patient_to_dicom[patient_id]\n            processed_img = preprocess_dicom_path(dicom_path)\n            image_data_list.append(processed_img)\n            processed_count += 1\n        else:\n            # If a patient ID is in the CSV but had no DICOMs found, append a zero array\n            image_data_list.append(np.zeros((IMG_SIZE, IMG_SIZE, 1), dtype=np.float32))\n\n    # Convert the list of arrays to a single NumPy array\n    X_img_data = np.array(image_data_list)\n    \n    if X_img_data.shape[0] != NUM_SAMPLES:\n        # This occurs if some patients were entirely missing, but we added zero arrays\n        # The data alignment should be correct now due to the list appends.\n        print(f\"Warning: Image data length ({X_img_data.shape[0]}) does not match tabular length ({NUM_SAMPLES}).\")\n\n    print(f\"Data prepared: {NUM_SAMPLES} samples. {processed_count} images successfully loaded.\")\n    return X_img_data, X_tab_data, y_data\n\n# ==============================================================================\n# 5. MAIN EXECUTION (Training and Saving)\n# ==============================================================================\n\nif __name__ == '__main__':\n    print(f\"TensorFlow Version: {tf.__version__}\")\n    print(\"Building EfficientNet-B3 Dual Input model...\")\n    model = build_model()\n    \n    print(\"\\n--- LOADING AND PREPARING DATA ---\")\n    \n    try:\n        X_img_data, X_tab_data, y_data = load_data(\n            dicom_dir=KAGGLE_DICOM_DIR,\n            csv_path=KAGGLE_CSV_PATH\n        )\n        \n        # Handle cases where image loading failed (resulting in zero arrays)\n        if X_img_data.shape[0] == 0:\n            raise ValueError(\"No data loaded. Check DICOM paths.\")\n\n        # Split data for training and validation\n        X_img_train, X_img_val, X_tab_train, X_tab_val, y_train, y_val = train_test_split(\n            X_img_data, X_tab_data, y_data, test_size=0.2, random_state=42\n        )\n        \n        print(f\"Dataset Split: Training ({len(y_train)}), Validation ({len(y_val)})\")\n        print(\"------------------------------------------------------------------\")\n\n    except Exception as e:\n        print(f\"\\n❌ FATAL ERROR in Data Loading: {e}\")\n        print(\"Model structure will be saved with dummy weights.\")\n        \n        # Fallback to small dummy arrays to save model structure only\n        X_img_train = np.random.rand(4, IMG_SIZE, IMG_SIZE, 1).astype(np.float32)\n        X_tab_train = np.random.rand(4, TABULAR_DIM).astype(np.float32)\n        y_train = np.random.rand(4, 2).astype(np.float32)\n        X_img_val = X_img_train\n        X_tab_val = X_tab_train\n        y_val = y_train\n\n\n    # --- MODEL TRAINING ---\n    print(f\"Starting training for {EPOCHS} epochs with batch size {BATCH_SIZE}...\")\n\n    try:\n        model.fit(\n            x=[X_img_train, X_tab_train],\n            y=y_train,\n            validation_data=([X_img_val, X_tab_val], y_val),\n            epochs=EPOCHS,\n            batch_size=BATCH_SIZE,\n            verbose=1 \n        )\n        \n        # --- SAVE MODEL WEIGHTS ---\n        model.save_weights(MODEL_WEIGHTS_PATH)\n        print(f\"\\n✅ SUCCESSFULLY SAVED WEIGHTS FILE AT: {MODEL_WEIGHTS_PATH}\")\n        print(\"\\n*** NEXT STEP: Download this file and push it to your GitHub for Vercel deployment. ***\")\n\n    except Exception as e:\n        print(f\"\\n❌ CRITICAL ERROR during training: {e}\")\n        print(\"Ensure you have TensorFlow and all required libraries installed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T05:51:53.930795Z","iopub.execute_input":"2025-11-16T05:51:53.931703Z","iopub.status.idle":"2025-11-16T05:56:19.973391Z","shell.execute_reply.started":"2025-11-16T05:51:53.931673Z","shell.execute_reply":"2025-11-16T05:56:19.972521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}