{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20604,"databundleVersionId":1357052,"sourceType":"competition"},{"sourceId":1139701,"sourceType":"datasetVersion","datasetId":642364},{"sourceId":1383977,"sourceType":"datasetVersion","datasetId":807680},{"sourceId":1401932,"sourceType":"datasetVersion","datasetId":774060},{"sourceId":1407014,"sourceType":"datasetVersion","datasetId":820718}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"hello OSIC\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:35:39.053862Z","iopub.execute_input":"2025-04-26T17:35:39.054104Z","iopub.status.idle":"2025-04-26T17:35:39.060883Z","shell.execute_reply.started":"2025-04-26T17:35:39.054086Z","shell.execute_reply":"2025-04-26T17:35:39.060106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# Clear GPU memory\ntf.keras.backend.clear_session()\ngpus = tf.config.experimental.list_physical_devices('GPU')\nif gpus:\n    try:\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n    except RuntimeError as e:\n        print(e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:35:42.016281Z","iopub.execute_input":"2025-04-26T17:35:42.017336Z","iopub.status.idle":"2025-04-26T17:35:55.965542Z","shell.execute_reply.started":"2025-04-26T17:35:42.017301Z","shell.execute_reply":"2025-04-26T17:35:55.964773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install efficientnet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:36:01.825669Z","iopub.execute_input":"2025-04-26T17:36:01.826621Z","iopub.status.idle":"2025-04-26T17:36:06.297983Z","shell.execute_reply.started":"2025-04-26T17:36:01.826594Z","shell.execute_reply":"2025-04-26T17:36:06.297217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\nimport os\nfrom glob import glob\nfrom scipy.ndimage import zoom, rotate\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Model, Input\nfrom tensorflow.keras.applications import ResNet50\nimport tensorflow.keras.backend as K\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:36:22.972853Z","iopub.execute_input":"2025-04-26T17:36:22.973549Z","iopub.status.idle":"2025-04-26T17:36:23.88436Z","shell.execute_reply.started":"2025-04-26T17:36:22.973513Z","shell.execute_reply":"2025-04-26T17:36:23.883792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Set random seed for reproducibility\nnp.random.seed(42)\ntf.random.set_seed(42)\n\n# Explicitly set float type to float32\nK.set_floatx('float32')\n\n# Define paths to the dataset\nBASE_PATH = \"/kaggle/input/osic-pulmonary-fibrosis-progression/\"\nTRAIN_DICOM_PATH = os.path.join(BASE_PATH, \"train\")\nTRAIN_CSV_PATH = os.path.join(BASE_PATH, \"train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:36:49.410012Z","iopub.execute_input":"2025-04-26T17:36:49.410967Z","iopub.status.idle":"2025-04-26T17:36:49.415362Z","shell.execute_reply.started":"2025-04-26T17:36:49.410939Z","shell.execute_reply":"2025-04-26T17:36:49.414589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# 1. Distribution des variables\nplt.figure(figsize=(15, 10))\n\n# Âge\nplt.subplot(2, 3, 1)\nsns.histplot(train_df['Age'], bins=20, kde=True, color='blue')\nplt.title('Distribution de l’âge')\nplt.xlabel('Âge')\nplt.ylabel('Nombre')\n\n# Sexe\nplt.subplot(2, 3, 2)\nsns.countplot(x='Sex', data=train_df, palette='Set2')\nplt.title('Répartition des sexes')\nplt.xlabel('Sexe')\nplt.ylabel('Nombre')\n\n# Statut de tabagisme\nplt.subplot(2, 3, 3)\nsns.countplot(x='SmokingStatus', data=train_df, palette='Set3')\nplt.title('Statut de tabagisme')\nplt.xlabel('Statut')\nplt.xticks(rotation=45)\nplt.ylabel('Nombre')\n\n# FVC\nplt.subplot(2, 3, 4)\nsns.histplot(train_df['FVC'], bins=20, kde=True, color='green')\nplt.title('Distribution de la FVC (mL)')\nplt.xlabel('FVC (mL)')\nplt.ylabel('Nombre')\n\n# Semaines\nplt.subplot(2, 3, 5)\nsns.histplot(train_df['Weeks'], bins=20, kde=True, color='purple')\nplt.title('Distribution des semaines')\nplt.xlabel('Semaines')\nplt.ylabel('Nombre')\n\nplt.tight_layout()\nplt.show()  # Afficher directement dans la cellule","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:15:03.535429Z","iopub.execute_input":"2025-04-26T19:15:03.535754Z","iopub.status.idle":"2025-04-26T19:15:05.011379Z","shell.execute_reply.started":"2025-04-26T19:15:03.535703Z","shell.execute_reply":"2025-04-26T19:15:05.010681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Heatmap des corrélations\nplt.figure(figsize=(8, 6))\nnumeric_cols = ['FVC', 'Weeks', 'Age', 'Percent']\ncorrelation_matrix = train_df[numeric_cols].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', vmin=-1, vmax=1)\nplt.title('Corrélation entre variables numériques')\nplt.show()  # Afficher directement dans la cellule","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:15:18.603317Z","iopub.execute_input":"2025-04-26T19:15:18.603873Z","iopub.status.idle":"2025-04-26T19:15:18.796617Z","shell.execute_reply.started":"2025-04-26T19:15:18.603849Z","shell.execute_reply":"2025-04-26T19:15:18.795974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Progression de la FVC pour quelques patients\nplt.figure(figsize=(10, 6))\nsample_patients = train_df['Patient'].unique()[:5]\nfor patient_id in sample_patients:\n    patient_data = train_df[train_df['Patient'] == patient_id]\n    plt.plot(patient_data['Weeks'], patient_data['FVC'], marker='o', label=f'Patient {patient_id[:5]}')\nplt.title('Progression de la FVC pour quelques patients')\nplt.xlabel('Semaines')\nplt.ylabel('FVC (mL)')\nplt.legend()\nplt.show()  # Afficher directement dans la cellule","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:15:34.194566Z","iopub.execute_input":"2025-04-26T19:15:34.194923Z","iopub.status.idle":"2025-04-26T19:15:34.385569Z","shell.execute_reply.started":"2025-04-26T19:15:34.194903Z","shell.execute_reply":"2025-04-26T19:15:34.384872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and preprocess DICOM data (unchanged)\ndef load_and_preprocess_dicom(folder_path, target_slices=36, target_resolution=(256, 256)):\n    try:\n        dicom_files = [os.path.join(folder_path, f) for f in os.listdir(folder_path) if f.endswith('.dcm')]\n        if not dicom_files:\n            raise ValueError(\"No DICOM files found\")\n        \n        slices = []\n        for file in dicom_files:\n            ds = pydicom.dcmread(file)\n            if hasattr(ds, 'InstanceNumber'):\n                slices.append(ds)\n        if not slices:\n            raise ValueError(\"No valid DICOM slices\")\n        slices.sort(key=lambda x: int(x.InstanceNumber))\n        \n        image_stack = np.stack([s.pixel_array for s in slices])\n        image_stack = np.clip(image_stack, -1024, 600)\n        image_stack = (image_stack - (-1024)) / (600 - (-1024) + 1e-10)\n        \n        min_dim = min(image_stack.shape[1], image_stack.shape[2])\n        h_start = (image_stack.shape[1] - min_dim) // 2\n        w_start = (image_stack.shape[2] - min_dim) // 2\n        image_stack = image_stack[:, h_start:h_start+min_dim, w_start:w_start+min_dim]\n        \n        resized_stack = []\n        for i in range(image_stack.shape[0]):\n            slice_img = image_stack[i]\n            if slice_img.shape != target_resolution:\n                slice_img = np.array(tf.image.resize(slice_img[..., np.newaxis], target_resolution)[..., 0])\n            resized_stack.append(slice_img)\n        image_stack = np.stack(resized_stack)\n        \n        current_slices = image_stack.shape[0]\n        if current_slices != target_slices:\n            if current_slices < target_slices:\n                padding = np.zeros((target_slices - current_slices, *target_resolution), dtype=np.float32)\n                image_stack = np.concatenate([image_stack, padding], axis=0)\n            else:\n                indices = np.linspace(0, current_slices - 1, target_slices).astype(int)\n                image_stack = image_stack[indices]\n        \n        assert image_stack.shape == (target_slices, target_resolution[0], target_resolution[1])\n        return image_stack.astype(np.float32)\n    except Exception as e:\n        raise e\n\n# Concatenate Tile Pooling (unchanged)\ndef concatenate_tile_pooling(image_stack, tiles_per_side=6):\n    T, H, W = image_stack.shape\n    assert T == tiles_per_side * tiles_per_side, f\"Expected {tiles_per_side * tiles_per_side} slices, got {T}\"\n    \n    tiled = image_stack.reshape(tiles_per_side, tiles_per_side, H, W)\n    tiled = tiled.transpose(0, 2, 1, 3)\n    tiled = tiled.reshape(tiles_per_side * H, tiles_per_side * W)\n    tiled = tiled[..., np.newaxis]\n    return tiled","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sélectionner un patient pour visualiser une image DICOM\npatient_id = train_df['Patient'].unique()[1]  # Premier patient\npatient_folder = os.path.join(TRAIN_DICOM_PATH, patient_id)\n# Charger une tranche DICOM brute\ndicom_files = [os.path.join(patient_folder, f) for f in os.listdir(patient_folder) if f.endswith('.dcm')]\ndicom_files.sort()  # Trier pour avoir un ordre cohérent\nds = pydicom.dcmread(dicom_files[len(dicom_files)//2])  # Prendre une tranche au milieu\nraw_image = ds.pixel_array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:16:39.101992Z","iopub.execute_input":"2025-04-26T19:16:39.1028Z","iopub.status.idle":"2025-04-26T19:16:39.128201Z","shell.execute_reply.started":"2025-04-26T19:16:39.102771Z","shell.execute_reply":"2025-04-26T19:16:39.127645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Appliquer le prétraitement\npreprocessed_stack = load_and_preprocess_dicom(patient_folder)\n# Prendre une tranche au milieu après prétraitement\npreprocessed_slice = preprocessed_stack[len(preprocessed_stack)//2]\n# Appliquer le tiling pour obtenir l'image finale\ntiled_image = concatenate_tile_pooling(preprocessed_stack)\n\n# Visualisation avant et après prétraitement\nplt.figure(figsize=(15, 5))\n\n# Image brute\nplt.subplot(1, 3, 1)\nplt.imshow(raw_image, cmap='gray')\nplt.title('Image DICOM brute')\nplt.axis('off')\n\n# Tranche après prétraitement (mais avant tiling)\nplt.subplot(1, 3, 2)\nplt.imshow(preprocessed_slice, cmap='gray')\nplt.title('Tranche après prétraitement')\nplt.axis('off')\n\n# Image après tiling\nplt.subplot(1, 3, 3)\nplt.imshow(tiled_image[..., 0], cmap='gray')\nplt.title('Image après tiling (1536x1536)')\nplt.axis('off')\n\nplt.tight_layout()\nplt.show()  # Afficher directement dans la cellule\n\nprint(\"Visualisations affichées dans la sortie de la cellule.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:16:55.30653Z","iopub.execute_input":"2025-04-26T19:16:55.306914Z","iopub.status.idle":"2025-04-26T19:17:05.561938Z","shell.execute_reply.started":"2025-04-26T19:16:55.306892Z","shell.execute_reply":"2025-04-26T19:17:05.561077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and preprocess DICOM data (unchanged)\ndef load_and_preprocess_dicom(folder_path, target_slices=36, target_resolution=(256, 256)):\n    try:\n        dicom_files = [os.path.join(folder_path, f) for f in os.listdir(folder_path) if f.endswith('.dcm')]\n        if not dicom_files:\n            raise ValueError(\"No DICOM files found\")\n        \n        slices = []\n        for file in dicom_files:\n            ds = pydicom.dcmread(file)\n            if hasattr(ds, 'InstanceNumber'):\n                slices.append(ds)\n        if not slices:\n            raise ValueError(\"No valid DICOM slices\")\n        slices.sort(key=lambda x: int(x.InstanceNumber))\n        \n        image_stack = np.stack([s.pixel_array for s in slices])\n        image_stack = np.clip(image_stack, -1024, 600)\n        image_stack = (image_stack - (-1024)) / (600 - (-1024) + 1e-10)\n        \n        min_dim = min(image_stack.shape[1], image_stack.shape[2])\n        h_start = (image_stack.shape[1] - min_dim) // 2\n        w_start = (image_stack.shape[2] - min_dim) // 2\n        image_stack = image_stack[:, h_start:h_start+min_dim, w_start:w_start+min_dim]\n        \n        resized_stack = []\n        for i in range(image_stack.shape[0]):\n            slice_img = image_stack[i]\n            if slice_img.shape != target_resolution:\n                slice_img = np.array(tf.image.resize(slice_img[..., np.newaxis], target_resolution)[..., 0])\n            resized_stack.append(slice_img)\n        image_stack = np.stack(resized_stack)\n        \n        current_slices = image_stack.shape[0]\n        if current_slices != target_slices:\n            if current_slices < target_slices:\n                padding = np.zeros((target_slices - current_slices, *target_resolution), dtype=np.float32)\n                image_stack = np.concatenate([image_stack, padding], axis=0)\n            else:\n                indices = np.linspace(0, current_slices - 1, target_slices).astype(int)\n                image_stack = image_stack[indices]\n        \n        assert image_stack.shape == (target_slices, target_resolution[0], target_resolution[1])\n        return image_stack.astype(np.float32)\n    except Exception as e:\n        raise e\n\n# Concatenate Tile Pooling (unchanged)\ndef concatenate_tile_pooling(image_stack, tiles_per_side=6):\n    T, H, W = image_stack.shape\n    assert T == tiles_per_side * tiles_per_side, f\"Expected {tiles_per_side * tiles_per_side} slices, got {T}\"\n    \n    tiled = image_stack.reshape(tiles_per_side, tiles_per_side, H, W)\n    tiled = tiled.transpose(0, 2, 1, 3)\n    tiled = tiled.reshape(tiles_per_side * H, tiles_per_side * W)\n    tiled = tiled[..., np.newaxis]\n    return tiled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:37:10.64052Z","iopub.execute_input":"2025-04-26T17:37:10.641066Z","iopub.status.idle":"2025-04-26T17:37:10.651678Z","shell.execute_reply.started":"2025-04-26T17:37:10.641045Z","shell.execute_reply":"2025-04-26T17:37:10.651019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enhanced Data Augmentation with Shear\ndef augment_2d_image(image, max_angle=30, scale_range=(0.7, 1.3), brightness_delta=0.3, contrast_range=(0.7, 1.3), shear_range=(-0.2, 0.2)):\n    image = image[..., 0]\n    \n    # Rotation\n    angle = np.random.uniform(-max_angle, max_angle)\n    image = rotate(image, angle, reshape=False, order=1)\n    \n    # Flipping\n    if np.random.rand() > 0.5:\n        image = np.flip(image, axis=0)\n    if np.random.rand() > 0.5:\n        image = np.flip(image, axis=1)\n    \n    # Scaling\n    scale_factor = np.random.uniform(scale_range[0], scale_range[1])\n    original_shape = image.shape\n    image = zoom(image, scale_factor, order=1)\n    resize_factor = [original_shape[i] / image.shape[i] for i in range(2)]\n    image = zoom(image, resize_factor, order=1)\n    if image.shape != original_shape:\n        image = np.resize(image, original_shape)\n    \n    # Shear\n    shear = np.random.uniform(shear_range[0], shear_range[1])\n    shear_matrix = np.array([[1, shear], [0, 1]])\n    image = tf.keras.preprocessing.image.apply_affine_transform(\n        image[..., np.newaxis], shear=shear*180/np.pi, fill_mode='nearest'\n    )[..., 0]\n    \n    # Brightness\n    brightness = np.random.uniform(-brightness_delta, brightness_delta)\n    image = image + brightness\n    \n    # Contrast\n    contrast = np.random.uniform(contrast_range[0], contrast_range[1])\n    image = (image - 0.5) * contrast + 0.5\n    \n    image = np.clip(image, 0, 1)\n    image = image[..., np.newaxis]\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:37:30.767951Z","iopub.execute_input":"2025-04-26T17:37:30.768772Z","iopub.status.idle":"2025-04-26T17:37:30.776102Z","shell.execute_reply.started":"2025-04-26T17:37:30.768713Z","shell.execute_reply":"2025-04-26T17:37:30.775311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load metadata and process DICOM data (unchanged)\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\nimage_data = {}\nsuccessful_patient_ids = []\n\nfor patient_id in train_df['Patient'].unique():\n    folder = os.path.join(TRAIN_DICOM_PATH, patient_id)\n    print(f\"Processing patient: {patient_id}\")\n    try:\n        image_data[patient_id] = load_and_preprocess_dicom(folder)\n        successful_patient_ids.append(patient_id)\n    except Exception as e:\n        print(f\"Failed to process patient {patient_id}: {str(e)}\")\n        continue\n\ntrain_df = train_df[train_df['Patient'].isin(successful_patient_ids)]\nprint(f\"Successfully processed {len(successful_patient_ids)} patients\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:37:42.118653Z","iopub.execute_input":"2025-04-26T17:37:42.119434Z","iopub.status.idle":"2025-04-26T17:50:28.551225Z","shell.execute_reply.started":"2025-04-26T17:37:42.119403Z","shell.execute_reply":"2025-04-26T17:50:28.550554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute slope 'a' for each patient (unchanged)\nA = {}\nfor p in train_df['Patient'].unique():\n    sub = train_df[train_df['Patient'] == p]\n    fvc = sub['FVC'].values\n    weeks = sub['Weeks'].values\n    c = np.vstack([weeks, np.ones(len(weeks))]).T\n    a, b = np.linalg.lstsq(c, fvc, rcond=None)[0]\n    A[p] = a\n\n# Prepare training and validation data (unchanged)\nimages = []\nmetadata = []\nslopes = []\n\nfor patient_id in train_df['Patient'].unique():\n    if patient_id in image_data:\n        image = image_data[patient_id]\n        assert image.shape == (36, 256, 256)\n        image = concatenate_tile_pooling(image)\n        images.append(image)\n        patient_data = train_df[train_df['Patient'] == patient_id].iloc[0]\n        metadata.append([\n            patient_data['Weeks'],\n            patient_data['Age'],\n            1 if patient_data['Sex'] == 'Male' else 0,\n            1 if patient_data['SmokingStatus'] == 'Ex-smoker' else 0,\n            1 if patient_data['SmokingStatus'] == 'Never smoked' else 0,\n            1 if patient_data['SmokingStatus'] == 'Currently smokes' else 0\n        ])\n        slopes.append(A[patient_id])\n\nimages = np.array(images, dtype=np.float32)\nmetadata = np.array(metadata, dtype=np.float32)\nslopes = np.array(slopes, dtype=np.float32)\n\n# Normalize slopes for training (unchanged)\nslope_min = slopes.min()\nslope_max = slopes.max()\nslopes_normalized = (slopes - slope_min) / (slope_max - slope_min)\n\n# Stratified split (unchanged)\nfvc_values = train_df.groupby('Patient')['FVC'].first().values\nnum_bins = 3\nfvc_bins = pd.cut(fvc_values, bins=num_bins, labels=False)\nX_train_idx, X_val_idx = train_test_split(\n    np.arange(len(images)),\n    test_size=0.2,\n    stratify=fvc_bins,\n    random_state=42\n)\n\nX_images_train = images[X_train_idx]\nX_images_val = images[X_val_idx]\nX_metadata_train = metadata[X_train_idx]\nX_metadata_val = metadata[X_val_idx]\ny_train = slopes_normalized[X_train_idx]\ny_val = slopes_normalized[X_val_idx]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:53:30.107138Z","iopub.execute_input":"2025-04-26T17:53:30.107437Z","iopub.status.idle":"2025-04-26T17:53:31.996823Z","shell.execute_reply.started":"2025-04-26T17:53:30.107416Z","shell.execute_reply":"2025-04-26T17:53:31.996222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply enhanced data augmentation to training images\nX_images_train = np.array([augment_2d_image(img) for img in X_images_train])\nprint(\"Applied enhanced data augmentation to training images\")\n\n# Scale metadata (unchanged)\nscaler = StandardScaler()\nX_metadata_train = scaler.fit_transform(X_metadata_train).astype(np.float32)\nX_metadata_val = scaler.transform(X_metadata_val).astype(np.float32)\n\nprint(\"Training samples:\", len(X_train_idx))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:53:34.401097Z","iopub.execute_input":"2025-04-26T17:53:34.401785Z","iopub.status.idle":"2025-04-26T17:54:37.081817Z","shell.execute_reply.started":"2025-04-26T17:53:34.401757Z","shell.execute_reply":"2025-04-26T17:54:37.081021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build the model (unchanged)\nimage_input = Input(shape=(1536, 1536, 1), name='image_input')\nx = layers.Concatenate(axis=-1)([image_input, image_input, image_input])\nbase_model = tf.keras.applications.ResNet50(\n    include_top=False,\n    weights='imagenet',\n    input_shape=(1536, 1536, 3)\n)\nfor layer in base_model.layers[50:]:\n    layer.trainable = False\nx = base_model(x, training=True)\nx = layers.GlobalAveragePooling2D()(x)\n\nmetadata_input = Input(shape=(6,), name='metadata_input')\nm = layers.Dense(64, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(0.02))(metadata_input)\nm = layers.Dense(32, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(0.02))(m)\nm = layers.Dropout(0.6)(m)\n\ncombined = layers.concatenate([x, m])\nz = layers.Dense(128, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(0.02))(combined)\nz = layers.Dropout(0.6)(z)\n\nslope_output = layers.Dense(1, activation='sigmoid', name='slope_output')(z)\n\nmodel = Model(\n    inputs=[image_input, metadata_input],\n    outputs=slope_output\n)\n\nprint(\"Model Output Shape:\", model.output_shape)\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:54:40.109788Z","iopub.execute_input":"2025-04-26T17:54:40.110458Z","iopub.status.idle":"2025-04-26T17:54:47.331126Z","shell.execute_reply.started":"2025-04-26T17:54:40.110432Z","shell.execute_reply":"2025-04-26T17:54:47.330528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Custom loss and metrics (unchanged)\ndef mae_loss(y_true, y_pred):\n    return tf.reduce_mean(tf.abs(y_true - y_pred))\n\ndef mae_in_ml(y_true, y_pred, batch_patient_ids):\n    slope_pred = tf.reshape(y_pred, [-1])\n    slope_true = tf.reshape(y_true, [-1])\n    slope_pred_denorm = slope_pred * (slope_max - slope_min) + slope_min\n    slope_true_denorm = slope_true * (slope_max - slope_min) + slope_min\n    \n    fvc_pred_list = []\n    fvc_true_list = []\n    for idx, pid in enumerate(batch_patient_ids):\n        patient_data = train_df[train_df['Patient'] == pid].iloc[0]\n        weeks = patient_data['Weeks']\n        fvc_base = patient_data['FVC']\n        fvc_pred = fvc_base + slope_pred_denorm[idx] * weeks\n        fvc_true = fvc_base + slope_true_denorm[idx] * weeks\n        fvc_pred_list.append(fvc_pred)\n        fvc_true_list.append(fvc_true)\n    \n    fvc_pred_list = tf.stack(fvc_pred_list)\n    fvc_true_list = tf.stack(fvc_true_list)\n    return tf.reduce_mean(tf.abs(fvc_true_list - fvc_pred_list))\n\n# Gradient accumulation (unchanged)\ndef gradient_accumulation_step(model, inputs, targets, optimizer, accum_steps=2):\n    total_gradients = None\n    total_loss = 0\n    \n    for i in range(accum_steps):\n        start_idx = i * (len(inputs[0]) // accum_steps)\n        end_idx = (i + 1) * (len(inputs[0]) // accum_steps)\n        batch_inputs = [inp[start_idx:end_idx] for inp in inputs]\n        batch_targets = targets[start_idx:end_idx]\n        \n        with tf.GradientTape() as tape:\n            predictions = model(batch_inputs, training=True)\n            loss = mae_loss(batch_targets, predictions)\n        \n        gradients = tape.gradient(loss, model.trainable_variables)\n        total_loss += loss\n        \n        if total_gradients is None:\n            total_gradients = gradients\n        else:\n            total_gradients = [g1 + g2 for g1, g2 in zip(total_gradients, gradients)]\n    \n    total_gradients = [g / accum_steps for g in total_gradients]\n    optimizer.apply_gradients(zip(total_gradients, model.trainable_variables))\n    return total_loss / accum_steps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:55:10.007633Z","iopub.execute_input":"2025-04-26T17:55:10.007944Z","iopub.status.idle":"2025-04-26T17:55:10.016954Z","shell.execute_reply.started":"2025-04-26T17:55:10.007921Z","shell.execute_reply":"2025-04-26T17:55:10.016123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add label noise\ndef add_label_noise(targets, noise_factor=0.02):\n    noise = np.random.normal(0, noise_factor, size=targets.shape)\n    noisy_targets = targets + noise\n    noisy_targets = np.clip(noisy_targets, 0, 1)\n    return noisy_targets.astype(np.float32)\n\n# Apply label noise to training targets\ny_train_noisy = add_label_noise(y_train, noise_factor=0.02)\nprint(\"Added label noise to training targets\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:55:21.155642Z","iopub.execute_input":"2025-04-26T17:55:21.155975Z","iopub.status.idle":"2025-04-26T17:55:21.161348Z","shell.execute_reply.started":"2025-04-26T17:55:21.155954Z","shell.execute_reply":"2025-04-26T17:55:21.160641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Update model regularization\nfor layer in model.layers:\n    if isinstance(layer, tf.keras.layers.Dense):\n        layer.kernel_regularizer = tf.keras.regularizers.l2(0.05)  # Increased to 0.05\nfor layer in model.layers:\n    if isinstance(layer, tf.keras.layers.Dropout):\n        layer.rate = 0.8  # Increased to 0.8\n\n# Compile model\noptimizer = tf.keras.optimizers.Adam(learning_rate=1e-5, clipnorm=0.5)\nmodel.compile(\n    optimizer=optimizer,\n    loss=mae_loss,\n    metrics=[mae_in_ml]\n)\n\n# Custom training loop with updated parameters\nepochs = 10\naccum_steps = 2\nsteps_per_epoch = len(X_train_idx) // (accum_steps * 1)\nval_steps = len(X_val_idx) // 1\n\nlr_factor = 0.5  # Increased for more aggressive reduction\nlr_patience = 1\nmin_lr = 1e-6\nbest_mae = float('inf')\nepochs_without_improvement_lr = 0\n\nes_patience = 3  # Reduced for tighter early stopping\nepochs_without_improvement_es = 0\nbest_weights = None\n\nbest_mae_for_checkpoint = float('inf')\nhistory = {'mae_in_ml': [], 'val_mae_in_ml': []}\n\nall_patient_ids = train_df['Patient'].unique()\n\nfor epoch in range(epochs):\n    print(f\"\\nEpoch {epoch+1}/{epochs}\")\n    \n    train_loss = 0\n    train_mae = 0\n    with tqdm(total=steps_per_epoch, desc=\"Training\", unit=\"step\") as pbar:\n        for step in range(steps_per_epoch):\n            start_idx = step * (accum_steps * 1)\n            end_idx = (step + 1) * (accum_steps * 1)\n            batch_inputs = [X_images_train[start_idx:end_idx], X_metadata_train[start_idx:end_idx]]\n            batch_targets = y_train_noisy[start_idx:end_idx]  # Use noisy targets\n            batch_indices = X_train_idx[start_idx:end_idx]\n            batch_patient_ids = all_patient_ids[batch_indices]\n            \n            loss = gradient_accumulation_step(model, batch_inputs, batch_targets, optimizer, accum_steps)\n            train_loss += loss\n            \n            predictions = model(batch_inputs, training=False)\n            mae = mae_in_ml(y_train[start_idx:end_idx], predictions, batch_patient_ids)  # Use original targets for metrics\n            train_mae += mae\n            \n            running_train_loss = train_loss / (step + 1)\n            running_train_mae = train_mae / (step + 1)\n            \n            pbar.set_postfix({\n                'train_loss': f\"{running_train_loss:.4f}\",\n                'train_mae_in_ml': f\"{running_train_mae:.2f}\"\n            })\n            pbar.update(1)\n    \n    train_loss /= steps_per_epoch\n    train_mae /= steps_per_epoch\n    \n    val_loss = 0\n    val_mae = 0\n    with tqdm(total=val_steps, desc=\"Validation\", unit=\"step\") as pbar:\n        for step in range(val_steps):\n            start_idx = step * 1\n            end_idx = (step + 1) * 1\n            batch_inputs = [X_images_val[start_idx:end_idx], X_metadata_val[start_idx:end_idx]]\n            batch_targets = y_val[start_idx:end_idx]\n            batch_indices = X_val_idx[start_idx:end_idx]\n            batch_patient_ids = all_patient_ids[batch_indices]\n            \n            predictions = model(batch_inputs, training=False)\n            loss = mae_loss(batch_targets, predictions)\n            mae = mae_in_ml(batch_targets, predictions, batch_patient_ids)\n            val_loss += loss\n            val_mae += mae\n            \n            running_val_loss = val_loss / (step + 1)\n            running_val_mae = val_mae / (step + 1)\n            \n            pbar.set_postfix({\n                'val_loss': f\"{running_val_loss:.4f}\",\n                'val_mae_in_ml': f\"{running_val_mae:.2f}\"\n            })\n            pbar.update(1)\n    \n    val_loss /= val_steps\n    val_mae /= val_steps\n    \n    history['mae_in_ml'].append(float(train_mae))\n    history['val_mae_in_ml'].append(float(val_mae))\n    \n    print(f\"Final metrics for Epoch {epoch+1}: train_loss: {train_loss:.4f} - train_mae_in_ml: {train_mae:.2f} - val_loss: {val_loss:.4f} - val_mae_in_ml: {val_mae:.2f}\")\n    \n    if train_mae < best_mae_for_checkpoint:\n        best_mae_for_checkpoint = train_mae\n        best_weights = model.get_weights()\n        model.save_weights('best_model.weights.h5')\n        print(f\"Epoch {epoch+1}: mae_in_ml improved from {best_mae_for_checkpoint:.5f} to {train_mae:.5f}, saving model to best_model.weights.h5\")\n    \n    # Learning rate reduction based on val_mae_in_ml\n    if val_mae < best_mae:\n        best_mae = val_mae\n        epochs_without_improvement_lr = 0\n    else:\n        epochs_without_improvement_lr += 1\n    \n    if epochs_without_improvement_lr >= lr_patience:\n        old_lr = float(optimizer.learning_rate)\n        new_lr = max(old_lr * lr_factor, min_lr)\n        optimizer.learning_rate.assign(new_lr)\n        print(f\"Reducing learning rate from {old_lr:.6f} to {new_lr:.6f} after {lr_patience} epochs without improvement in val_mae_in_ml\")\n        epochs_without_improvement_lr = 0\n    \n    # Early stopping based on val_mae_in_ml\n    if val_mae < best_mae:\n        best_mae = val_mae\n        epochs_without_improvement_es = 0\n        best_weights = model.get_weights()\n    else:\n        epochs_without_improvement_es += 1\n    \n    if epochs_without_improvement_es >= es_patience:\n        print(f\"Early stopping triggered at epoch {epoch+1} after {es_patience} epochs without improvement in val_mae_in_ml\")\n        if best_weights is not None:\n            model.set_weights(best_weights)\n            print(\"Restored best weights from checkpoint\")\n        break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T17:55:49.719796Z","iopub.execute_input":"2025-04-26T17:55:49.720076Z","iopub.status.idle":"2025-04-26T18:01:38.756772Z","shell.execute_reply.started":"2025-04-26T17:55:49.720056Z","shell.execute_reply":"2025-04-26T18:01:38.756007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Evaluate with batched prediction\nbatch_size = 1\nval_predictions = []\nfor i in range(0, len(X_images_val), batch_size):\n    batch_inputs = [X_images_val[i:i+batch_size], X_metadata_val[i:i+batch_size]]\n    preds = model.predict(batch_inputs, batch_size=batch_size, verbose=0)\n    val_predictions.append(preds)\nval_predictions = np.concatenate(val_predictions, axis=0)\n\n# Denormalize predictions and compute final FVC values\nval_slopes_denorm = val_predictions * (slope_max - slope_min) + slope_min\nval_slopes_true_denorm = y_val * (slope_max - slope_min) + slope_min\n\nval_fvc_predictions = []\nval_fvc_true = []\npatient_ids = train_df['Patient'].unique()[X_val_idx]\nfor idx, pid in enumerate(patient_ids):\n    patient_data = train_df[train_df['Patient'] == pid].iloc[0]\n    weeks = patient_data['Weeks']\n    fvc_base = patient_data['FVC']\n    fvc_pred = fvc_base + val_slopes_denorm[idx] * weeks\n    fvc_true = fvc_base + val_slopes_true_denorm[idx] * weeks\n    val_fvc_predictions.append(fvc_pred)\n    val_fvc_true.append(fvc_true)\n\nval_fvc_predictions = np.array(val_fvc_predictions)\nval_fvc_true = np.array(val_fvc_true)\n\n# Compute final MAE for validation set\nfinal_val_mae = mae_in_ml(y_val, val_predictions, patient_ids)\nprint(f\"Final Validation MAE: {final_val_mae:.2f} mL\")\n\n# Plot training history\nplt.figure(figsize=(10, 5))\nplt.plot(history['mae_in_ml'], label='Training MAE (mL)')\nplt.plot(history['val_mae_in_ml'], label='Validation MAE (mL)')\nplt.title('Training MAE')\nplt.xlabel('Epoch')\nplt.ylabel('MAE (mL)')\nplt.legend()\nplt.savefig('training_metrics.png')\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T18:15:48.343228Z","iopub.execute_input":"2025-04-26T18:15:48.343957Z","iopub.status.idle":"2025-04-26T18:15:52.645267Z","shell.execute_reply.started":"2025-04-26T18:15:48.343928Z","shell.execute_reply":"2025-04-26T18:15:52.644571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reapply stronger regularization\nfor layer in model.layers:\n    if isinstance(layer, tf.keras.layers.Dense):\n        layer.kernel_regularizer = tf.keras.regularizers.l2(0.07)  # Increased to 0.07\nfor layer in model.layers:\n    if isinstance(layer, tf.keras.layers.Dropout):\n        layer.rate = 0.85  # Increased to 0.85\n\n# Increase label noise\ny_train_noisy = add_label_noise(y_train, noise_factor=0.03)\nprint(\"Increased label noise to 0.03\")\n\n# Recompile model\noptimizer = tf.keras.optimizers.Adam(learning_rate=1e-5, clipnorm=0.5)\nmodel.compile(\n    optimizer=optimizer,\n    loss=mae_loss,\n    metrics=[mae_in_ml]\n)\n\n# Custom training loop with adjusted parameters\nepochs = 5  # Reduced to focus on early performance\naccum_steps = 2\nsteps_per_epoch = len(X_train_idx) // (accum_steps * 1)\nval_steps = len(X_val_idx) // 1\n\nlr_factor = 0.2  # More aggressive reduction\nlr_patience = 1\nmin_lr = 1e-6\nbest_mae = 75.34  # Best validation MAE so far\nepochs_without_improvement_lr = 0\n\nes_patience = 5\nepochs_without_improvement_es = 0\nbest_weights = model.get_weights()\nbest_mae_for_checkpoint = 58.52\n\nhistory = {'mae_in_ml': [124.68, 62.71, 58.52], 'val_mae_in_ml': [96.37, 83.45, 75.34]}\n\nall_patient_ids = train_df['Patient'].unique()\n\nfor epoch in range(3, epochs):\n    print(f\"\\nEpoch {epoch+1}/{epochs}\")\n    \n    train_loss = 0\n    train_mae = 0\n    with tqdm(total=steps_per_epoch, desc=\"Training\", unit=\"step\") as pbar:\n        for step in range(steps_per_epoch):\n            start_idx = step * (accum_steps * 1)\n            end_idx = (step + 1) * (accum_steps * 1)\n            batch_inputs = [X_images_train[start_idx:end_idx], X_metadata_train[start_idx:end_idx]]\n            batch_targets = y_train_noisy[start_idx:end_idx]\n            batch_indices = X_train_idx[start_idx:end_idx]\n            batch_patient_ids = all_patient_ids[batch_indices]\n            \n            loss = gradient_accumulation_step(model, batch_inputs, batch_targets, optimizer, accum_steps)\n            train_loss += loss\n            \n            predictions = model(batch_inputs, training=False)\n            mae = mae_in_ml(y_train[start_idx:end_idx], predictions, batch_patient_ids)\n            train_mae += mae\n            \n            running_train_loss = train_loss / (step + 1)\n            running_train_mae = train_mae / (step + 1)\n            \n            pbar.set_postfix({\n                'train_loss': f\"{running_train_loss:.4f}\",\n                'train_mae_in_ml': f\"{running_train_mae:.2f}\"\n            })\n            pbar.update(1)\n    \n    train_loss /= steps_per_epoch\n    train_mae /= steps_per_epoch\n    \n    val_loss = 0\n    val_mae = 0\n    with tqdm(total=val_steps, desc=\"Validation\", unit=\"step\") as pbar:\n        for step in range(val_steps):\n            start_idx = step * 1\n            end_idx = (step + 1) * 1\n            batch_inputs = [X_images_val[start_idx:end_idx], X_metadata_val[start_idx:end_idx]]\n            batch_targets = y_val[start_idx:end_idx]\n            batch_indices = X_val_idx[start_idx:end_idx]\n            batch_patient_ids = all_patient_ids[batch_indices]\n            \n            predictions = model(batch_inputs, training=False)\n            loss = mae_loss(batch_targets, predictions)\n            mae = mae_in_ml(batch_targets, predictions, batch_patient_ids)\n            val_loss += loss\n            val_mae += mae\n            \n            running_val_loss = val_loss / (step + 1)\n            running_val_mae = val_mae / (step + 1)\n            \n            pbar.set_postfix({\n                'val_loss': f\"{running_val_loss:.4f}\",\n                'val_mae_in_ml': f\"{running_val_mae:.2f}\"\n            })\n            pbar.update(1)\n    \n    val_loss /= val_steps\n    val_mae /= val_steps\n    \n    history['mae_in_ml'].append(float(train_mae))\n    history['val_mae_in_ml'].append(float(val_mae))\n    \n    print(f\"Final metrics for Epoch {epoch+1}: train_loss: {train_loss:.4f} - train_mae_in_ml: {train_mae:.2f} - val_loss: {val_loss:.4f} - val_mae_in_ml: {val_mae:.2f}\")\n    \n    if train_mae < best_mae_for_checkpoint:\n        best_mae_for_checkpoint = train_mae\n        best_weights = model.get_weights()\n        model.save_weights('best_model.weights.h5')\n        print(f\"Epoch {epoch+1}: mae_in_ml improved from {best_mae_for_checkpoint:.5f} to {train_mae:.5f}, saving model to best_model.weights.h5\")\n    \n    if val_mae < best_mae:\n        best_mae = val_mae\n        epochs_without_improvement_lr = 0\n    else:\n        epochs_without_improvement_lr += 1\n    \n    if epochs_without_improvement_lr >= lr_patience:\n        old_lr = float(optimizer.learning_rate)\n        new_lr = max(old_lr * lr_factor, min_lr)\n        optimizer.learning_rate.assign(new_lr)\n        print(f\"Reducing learning rate from {old_lr:.6f} to {new_lr:.6f} after {lr_patience} epochs without improvement in val_mae_in_ml\")\n        epochs_without_improvement_lr = 0\n    \n    if val_mae < best_mae:\n        best_mae = val_mae\n        epochs_without_improvement_es = 0\n        best_weights = model.get_weights()\n    else:\n        epochs_without_improvement_es += 1\n    \n    if epochs_without_improvement_es >= es_patience:\n        print(f\"Early stopping triggered at epoch {epoch+1} after {es_patience} epochs without improvement in val_mae_in_ml\")\n        if best_weights is not None:\n            model.set_weights(best_weights)\n            print(\"Restored best weights from checkpoint\")\n        break\n\n# Evaluate with batched prediction\nbatch_size = 1\nval_predictions = []\nfor i in range(0, len(X_images_val), batch_size):\n    batch_inputs = [X_images_val[i:i+batch_size], X_metadata_val[i:i+batch_size]]\n    preds = model.predict(batch_inputs, batch_size=batch_size, verbose=0)\n    val_predictions.append(preds)\nval_predictions = np.concatenate(val_predictions, axis=0)\n\nval_slopes_denorm = val_predictions * (slope_max - slope_min) + slope_min\nval_slopes_true_denorm = y_val * (slope_max - slope_min) + slope_min\n\nval_fvc_predictions = []\nval_fvc_true = []\npatient_ids = train_df['Patient'].unique()[X_val_idx]\nfor idx, pid in enumerate(patient_ids):\n    patient_data = train_df[train_df['Patient'] == pid].iloc[0]\n    weeks = patient_data['Weeks']\n    fvc_base = patient_data['FVC']\n    fvc_pred = fvc_base + val_slopes_denorm[idx] * weeks\n    fvc_true = fvc_base + val_slopes_true_denorm[idx] * weeks\n    val_fvc_predictions.append(fvc_pred)\n    val_fvc_true.append(fvc_true)\n\nval_fvc_predictions = np.array(val_fvc_predictions)\nval_fvc_true = np.array(val_fvc_true)\n\nfinal_val_mae = mae_in_ml(y_val, val_predictions, patient_ids)\nprint(f\"Final Validation MAE: {final_val_mae:.2f} mL\")\n\nplt.figure(figsize=(10, 5))\nplt.plot(history['mae_in_ml'], label='Training MAE (mL)')\nplt.plot(history['val_mae_in_ml'], label='Validation MAE (mL)')\nplt.title('Training MAE')\nplt.xlabel('Epoch')\nplt.ylabel('MAE (mL)')\nplt.legend()\nplt.savefig('training_metrics.png')\nplt.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:04:36.188568Z","iopub.execute_input":"2025-04-26T19:04:36.188881Z","iopub.status.idle":"2025-04-26T19:08:30.041629Z","shell.execute_reply.started":"2025-04-26T19:04:36.188859Z","shell.execute_reply":"2025-04-26T19:08:30.040948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure plots are displayed inline in the notebook\n%matplotlib inline\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Evaluate with batched prediction\nbatch_size = 1\nval_predictions = []\nfor i in range(0, len(X_images_val), batch_size):\n    batch_inputs = [X_images_val[i:i+batch_size], X_metadata_val[i:i+batch_size]]\n    preds = model.predict(batch_inputs, batch_size=batch_size, verbose=0)\n    val_predictions.append(preds)\nval_predictions = np.concatenate(val_predictions, axis=0)\n\n# Denormalize predicted slopes\nval_slopes_denorm = val_predictions * (slope_max - slope_min) + slope_min\n\n# Get patient IDs for the validation set\npatient_ids = train_df['Patient'].unique()[X_val_idx]\n\n# Compute predicted and true FVC values\nval_fvc_predictions = []\nval_fvc_true = []\n\nfor idx, pid in enumerate(patient_ids):\n    # Get patient data\n    patient_data = train_df[train_df['Patient'] == pid]\n    \n    # Use the first measurement as the baseline\n    baseline_data = patient_data.iloc[0]\n    weeks = baseline_data['Weeks']\n    fvc_base = baseline_data['FVC']\n    \n    # Predicted FVC using the denormalized slope\n    fvc_pred = fvc_base + val_slopes_denorm[idx] * weeks\n    val_fvc_predictions.append(fvc_pred)\n    \n    # True FVC using the true slope\n    true_slope = A[pid]  # True slope from earlier computation\n    fvc_true = fvc_base + true_slope * weeks\n    val_fvc_true.append(fvc_true)\n\nval_fvc_predictions = np.array(val_fvc_predictions)\nval_fvc_true = np.array(val_fvc_true)\n\n# Compute evaluation metrics\nmae = np.mean(np.abs(val_fvc_true - val_fvc_predictions))\nmse = mean_squared_error(val_fvc_true, val_fvc_predictions)\nr2 = r2_score(val_fvc_true, val_fvc_predictions)\n\nprint(f\"Evaluation on Real FVC Targets:\")\nprint(f\"Mean Absolute Error (MAE): {mae:.2f} mL\")\nprint(f\"Mean Squared Error (MSE): {mse:.2f} mL^2\")\nprint(f\"R-squared (R²): {r2:.4f}\")\n\n# Scatter plot of predicted vs. true FVC values (display inline)\nplt.figure(figsize=(8, 6))\nplt.scatter(val_fvc_true, val_fvc_predictions, alpha=0.5)\nplt.plot([val_fvc_true.min(), val_fvc_true.max()], [val_fvc_true.min(), val_fvc_true.max()], 'r--', lw=2)\nplt.xlabel('True FVC (mL)')\nplt.ylabel('Predicted FVC (mL)')\nplt.title('Predicted vs. True FVC Values')\nplt.grid(True)\nplt.show()\n\n# Optionally, display the training history plot inline as well\nplt.figure(figsize=(10, 5))\nplt.plot(history['mae_in_ml'], label='Training MAE (mL)')\nplt.plot(history['val_mae_in_ml'], label='Validation MAE (mL)')\nplt.title('Training MAE')\nplt.xlabel('Epoch')\nplt.ylabel('MAE (mL)')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:08:30.04285Z","iopub.execute_input":"2025-04-26T19:08:30.043103Z","iopub.status.idle":"2025-04-26T19:08:34.407312Z","shell.execute_reply.started":"2025-04-26T19:08:30.043084Z","shell.execute_reply":"2025-04-26T19:08:34.406671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure plots are displayed inline\n%matplotlib inline\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Evaluate with batched prediction\nbatch_size = 1\nval_predictions = []\nfor i in range(0, len(X_images_val), batch_size):\n    batch_inputs = [X_images_val[i:i+batch_size], X_metadata_val[i:i+batch_size]]\n    preds = model.predict(batch_inputs, batch_size=batch_size, verbose=0)\n    val_predictions.append(preds)\nval_predictions = np.concatenate(val_predictions, axis=0)\n\n# Denormalize predicted slopes\nval_slopes_denorm = val_predictions * (slope_max - slope_min) + slope_min\n\n# Get patient IDs for the validation set\npatient_ids = train_df['Patient'].unique()[X_val_idx]\n\n# Compute predicted and true FVC values over 12 weeks\nweeks_range = np.arange(0, 13)  # Weeks 0 to 12 inclusive\nall_predicted_fvc = []\nall_true_fvc = []\nall_abs_errors = []\n\nfor idx, pid in enumerate(patient_ids):\n    # Get patient data\n    patient_data = train_df[train_df['Patient'] == pid]\n    \n    # Use the first measurement as the baseline\n    baseline_data = patient_data.iloc[0]\n    fvc_base = baseline_data['FVC']\n    \n    # Predicted FVC over 12 weeks using the denormalized slope\n    predicted_slope = val_slopes_denorm[idx]\n    predicted_fvc = fvc_base + predicted_slope * weeks_range\n    all_predicted_fvc.append(predicted_fvc)\n    \n    # True FVC over 12 weeks using the true slope\n    true_slope = A[pid]\n    true_fvc = fvc_base + true_slope * weeks_range\n    all_true_fvc.append(true_fvc)\n    \n    # Compute absolute errors for this patient across all weeks\n    abs_errors = np.abs(true_fvc - predicted_fvc)\n    all_abs_errors.extend(abs_errors)\n\nall_predicted_fvc = np.array(all_predicted_fvc)  # Shape: (num_patients, 13)\nall_true_fvc = np.array(all_true_fvc)  # Shape: (num_patients, 13)\nall_abs_errors = np.array(all_abs_errors)\n\n# Compute MAE over the 12-week period\nmae_over_12_weeks = np.mean(all_abs_errors)\nprint(f\"MAE Over 12 Weeks (All Patients, All Weeks): {mae_over_12_weeks:.2f} mL\")\n\n# Outlier analysis at week 12\nfvc_true_week_12 = all_true_fvc[:, -1]\nfvc_pred_week_12 = all_predicted_fvc[:, -1]\nabs_errors_week_12 = np.abs(fvc_true_week_12 - fvc_pred_week_12)\ntop_errors_idx = np.argsort(abs_errors_week_12)[-5:][::-1]\nprint(\"\\nTop 5 Largest Errors at Week 12:\")\nfor idx in top_errors_idx:\n    print(f\"Patient {patient_ids[idx]}: True FVC = {fvc_true_week_12[idx]:.2f} mL, Predicted FVC = {fvc_pred_week_12[idx]:.2f} mL, Error = {abs_errors_week_12[idx]:.2f} mL\")\n\nnon_outlier_idx = np.ones(len(abs_errors_week_12), dtype=bool)\nnon_outlier_idx[top_errors_idx] = False\nmae_excluding_outliers = np.mean(abs_errors_week_12[non_outlier_idx])\nprint(f\"\\nMAE at Week 12 Excluding Top 5 Outliers: {mae_excluding_outliers:.2f} mL\")\n\n# Plot predicted vs. true FVC trajectories for a subset of patients\nnum_patients_to_plot = 5\nplt.figure(figsize=(12, 8))\nfor idx in range(min(num_patients_to_plot, len(patient_ids))):\n    plt.plot(weeks_range, all_true_fvc[idx], label=f'Patient {patient_ids[idx]} (True)', linestyle='-', marker='o')\n    plt.plot(weeks_range, all_predicted_fvc[idx], label=f'Patient {patient_ids[idx]} (Predicted)', linestyle='--', marker='x')\nplt.xlabel('Weeks')\nplt.ylabel('FVC (mL)')\nplt.title('Predicted vs. True FVC Over 12 Weeks')\nplt.legend(bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n\n# Scatter plot of predicted vs. true FVC values at week 12\nplt.figure(figsize=(8, 6))\nplt.scatter(fvc_true_week_12, fvc_pred_week_12, alpha=0.5)\nplt.plot([fvc_true_week_12.min(), fvc_true_week_12.max()], [fvc_true_week_12.min(), fvc_pred_week_12.max()], 'r--', lw=2)\nplt.xlabel('True FVC at Week 12 (mL)')\nplt.ylabel('Predicted FVC at Week 12 (mL)')\nplt.title('Predicted vs. True FVC at Week 12')\nplt.grid(True)\nplt.show()\n\n# Training history plot\nplt.figure(figsize=(10, 5))\nplt.plot(history['mae_in_ml'], label='Training MAE (mL)')\nplt.plot(history['val_mae_in_ml'], label='Validation MAE (mL)')\nplt.title('Training MAE')\nplt.xlabel('Epoch')\nplt.ylabel('MAE (mL)')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T19:11:42.663271Z","iopub.execute_input":"2025-04-26T19:11:42.663905Z","iopub.status.idle":"2025-04-26T19:11:47.360688Z","shell.execute_reply.started":"2025-04-26T19:11:42.663881Z","shell.execute_reply":"2025-04-26T19:11:47.359987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}