{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importation des librairies \n","metadata":{}},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 1. IMPORTS & SÉCURITÉ (SANS DEPENDANCE INTERNET)\n# ─────────────────────────────────────────────────────────────────────────────\nimport warnings\nwarnings.filterwarnings(action='ignore', category=FutureWarning)\n\n\nimport os\nimport time\nimport random\nimport cv2\nimport pydicom\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow.keras.backend as K        \nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import (\n    Dense, Dropout, Input, GlobalAveragePooling2D,\n    Concatenate, GaussianNoise, BatchNormalization, Layer\n)\nfrom tensorflow.keras.utils import Sequence\nfrom sklearn.metrics import mean_absolute_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:37:48.251018Z","iopub.execute_input":"2026-08-04T14:37:48.251492Z","iopub.status.idle":"2026-08-04T14:38:05.834393Z","shell.execute_reply.started":"2026-08-04T14:37:48.251461Z","shell.execute_reply":"2026-08-04T14:38:05.833758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 2. REPRODUCTIBILITÉ & GPU\n# ─────────────────────────────────────────────────────────────────────────────\ndef seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n \nseed_everything(42) # Mise en place du GPU \n \n# Permet à TensorFlow d'allouer la mémoire GPU progressivement\nconfig = tf.compat.v1.ConfigProto()\nconfig.gpu_options.allow_growth = True\nsession = tf.compat.v1.Session(config=config)\n \nROOT = \"../input/osic-pulmonary-fibrosis-progression\"\nBATCH_SIZE_IMG = 32    # Pour les modèles images (ConvNeXt)\nBATCH_SIZE_TAB = 32    # Pour le modèle tabulaire (était 128, trop grand)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:38:05.835884Z","iopub.execute_input":"2026-08-04T14:38:05.836359Z","iopub.status.idle":"2026-08-04T14:38:07.505573Z","shell.execute_reply.started":"2026-08-04T14:38:05.836336Z","shell.execute_reply":"2026-08-04T14:38:07.504674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 3. CHARGEMENT ET PRÉPARATION DES DONNÉES\n# ─────────────────────────────────────────────────────────────────────────────\ntrain = pd.read_csv(f'{ROOT}/train.csv') # Permet de lire le fichier\n \ndef get_tab(df):\n    \"\"\"Encode les features tabulaires d'un patient en vecteur numérique.\"\"\"\n    vector = [(df.Age.values[0] - 30) / 30]\n \n    vector.append(0 if df.Sex.values[0] == 'male' else 1)\n \n    smoking = df.SmokingStatus.values[0]\n    smoking_map = {\n        'Never smoked':      [0, 0],\n        'Ex-smoker':         [1, 1],\n        'Currently smokes':  [0, 1],\n    }\n    vector.extend(smoking_map.get(smoking, [1, 0]))\n    return np.array(vector)\n \n \n# Chargement des IRM pour redimensionner les IRM en 512x512.\ndef get_img(path):\n    d = pydicom.dcmread(path)\n    return cv2.resize(d.pixel_array / 2**11, (512, 512))\n \n \nA, TAB, P = {}, {}, []\nfor p in tqdm(train.Patient.unique(), desc=\"Régression par patient\"):\n    sub = train.loc[train.Patient == p, :]\n    fvc, weeks = sub.FVC.values, sub.Weeks.values\n    c = np.vstack([weeks, np.ones(len(weeks))]).T\n    a, b = np.linalg.lstsq(c, fvc, rcond=None)[0]\n    A[p] = a\n    TAB[p] = get_tab(sub)\n    P.append(p)\n \nsns.histplot(list(A.values()), kde=True)\nplt.title(\"Distribution des pentes de déclin du FVC\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:38:07.506505Z","iopub.execute_input":"2026-08-04T14:38:07.506805Z","iopub.status.idle":"2026-08-04T14:38:07.959035Z","shell.execute_reply.started":"2026-08-04T14:38:07.506781Z","shell.execute_reply":"2026-08-04T14:38:07.958355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 4. GÉNÉRATEUR D'IMAGES\n# ─────────────────────────────────────────────────────────────────────────────\nclass IGenerator(Sequence):\n    BAD_ID = ['ID00011637202177653955184', 'ID00052637202186188008618']\n \n    def __init__(self, keys, a, tab, batch_size=32):\n        self.keys = [k for k in keys if k not in self.BAD_ID]\n        self.a = a\n        self.tab = tab\n        self.batch_size = batch_size\n        self.train_data = {\n            p: os.listdir(f'{ROOT}/train/{p}/')\n            for p in train.Patient.values\n        }\n \n    def __len__(self):\n        return 1000\n \n    def __getitem__(self, idx):\n        x, a, tab = [], [], []\n        keys = np.random.choice(self.keys, size=self.batch_size)\n        for k in keys:\n            try:\n                i = np.random.choice(self.train_data[k], size=1)[0]\n                img = get_img(f'{ROOT}/train/{k}/{i}')\n                x.append(img)\n                a.append(self.a[k])\n                tab.append(self.tab[k])\n            except Exception as e:\n                print(f\"Erreur patient {k}, image {i}: {e}\")\n \n        x, a, tab = np.array(x), np.array(a), np.array(tab)\n        # CHANGEMENT : ConvNeXt attend 3 canaux (RGB) — on duplique le canal\n        # gris en 3 canaux au lieu de faire un simple expand_dims(-1).\n        x = np.repeat(x[..., np.newaxis], 3, axis=-1)\n        return [x, tab], a","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:38:07.959941Z","iopub.execute_input":"2026-08-04T14:38:07.960214Z","iopub.status.idle":"2026-08-04T14:38:07.968219Z","shell.execute_reply.started":"2026-08-04T14:38:07.960180Z","shell.execute_reply":"2026-08-04T14:38:07.967449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 5. MODÈLE IMAGE (ConvNeXt-Tiny + Features tabulaires)\n# ─────────────────────────────────────────────────────────────────────────────\n \ndef get_convnext_backbone(shape):\n    return tf.keras.applications.ConvNeXtTiny(\n        include_top=False,\n        weights=None, # ou 'imagenet' si tu as accès à Internet\n        input_shape=shape\n    )\n \n \ndef build_model(shape=(512, 512, 3)):\n    inp_img = Input(shape=shape, name=\"input_image\")\n    base = get_convnext_backbone(shape)\n    x = base(inp_img)\n    x = GlobalAveragePooling2D()(x)\n \n    inp_tab = Input(shape=(4,), name=\"input_tab\")\n    x2 = tf.keras.layers.GaussianNoise(0.2)(inp_tab)\n \n    x = Concatenate()([x, x2])\n    x = Dropout(0.2)(x)\n    x = Dense(1, name=\"output_pente\")(x)\n \n    model = Model([inp_img, inp_tab], x)\n    # Plus besoin de charger des poids externes par nom de couche (skip_mismatch) :\n    # les poids ImageNet sont déjà chargés nativement dans le backbone ci-dessus.\n    return model\n \n \n# CHANGEMENT : un seul modèle ConvNeXt-Tiny (plus de boucle sur plusieurs tailles b0-b7)\nmodels = [build_model(shape=(512, 512, 3))]\nprint(f'Nombre de modèles image chargés : {len(models)}')\n \ntr_p, vl_p = train_test_split(P, shuffle=True, train_size=0.8, random_state=42)\n \n \ndef score(fvc_true, fvc_pred, sigma):\n    sigma_clip = np.maximum(sigma, 75)\n    delta = np.minimum(np.abs(fvc_true - fvc_pred), 1000)\n    sq2 = np.sqrt(2)\n    metric = -(delta / sigma_clip) * sq2 - np.log(sigma_clip * sq2)\n    return np.mean(metric)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:38:07.969145Z","iopub.execute_input":"2026-08-04T14:38:07.969596Z","iopub.status.idle":"2026-08-04T14:38:09.985087Z","shell.execute_reply.started":"2026-08-04T14:38:07.969501Z","shell.execute_reply":"2026-08-04T14:38:09.983999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 8. INFÉRENCE MODÈLES IMAGE + GÉNÉRATION SOUMISSION IMAGE\n# ─────────────────────────────────────────────────────────────────────────────\ndef predict_fvc_from_images(model, patient_list, data_csv, data_dir, is_train=True):\n    P_FVC = {}\n    min_images = 1 if is_train else 2\n \n    try:\n        expected_channels = model.input_shape[0][-1]\n    except Exception:\n        expected_channels = 3\n \n    for p in tqdm(patient_list, desc=\"Inférence images\"):\n        if p in ['ID00011637202177653955184', 'ID00052637202186188008618']:\n            continue\n \n        ldir = os.listdir(f'{data_dir}/{p}/')\n        x, tab = [], []\n \n        for i in ldir:\n            ratio = int(i[:-4]) / len(ldir)\n            if 0.15 < ratio < 0.80:\n                x.append(get_img(f'{data_dir}/{p}/{i}'))\n                tab.append(get_tab(data_csv.loc[data_csv.Patient == p, :]))\n \n        if len(x) < min_images:\n            continue\n \n        x = np.array(x)\n        tab = np.array(tab)\n \n        # expected_channels vaudra 3 pour ConvNeXt -> ce bloc gère déjà\n        # correctement la duplication de canal, aucun changement nécessaire ici.\n        if expected_channels == 3:\n            x = np.repeat(x[..., np.newaxis], 3, axis=-1)\n        else:\n            x = np.expand_dims(x, axis=-1)\n \n        if len(tab.shape) == 1:\n            tab = np.expand_dims(tab, axis=0)\n \n        preds = model.predict([x, tab], verbose=0)\n \n        if len(preds.shape) > 1 and preds.shape[1] > 1:\n            P_FVC[p] = {\n                'pente': float(np.mean(preds[:, 0])),\n                'sigma_brut': float(np.mean(preds[:, 1]))\n            }\n        else:\n            P_FVC[p] = {\n                'pente': float(np.mean(preds)),\n                'sigma_brut': None\n            }\n \n    return P_FVC\n \n \ndef calibrate_best_sigma(fvc_true_list, fvc_pred_list, sigma_range=np.arange(70, 300, 5)):\n    best_score, best_sigma, best_std = -np.inf, sigma_range[0], None\n \n    for sigma_test in sigma_range:\n        scores_test = []\n        for fvc_true, fvc_pred in zip(fvc_true_list, fvc_pred_list):\n            conf_pred = np.full_like(fvc_pred, sigma_test, dtype=float)\n            scores_test.append(score(fvc_true, fvc_pred, conf_pred))\n        \n        mean_score = np.mean(scores_test)\n        if mean_score > best_score:\n            best_score, best_sigma = mean_score, sigma_test\n \n    return best_sigma, best_score\n \n \nsubs = []\nfor model in models:\n    train_csv = pd.read_csv(f'{ROOT}/train.csv')\n \n    P_FVC_val = predict_fvc_from_images(model, vl_p, train_csv, f'{ROOT}/train', is_train=True)\n \n    fvc_true_list, fvc_pred_list = [], []\n    for p, preds_dict in P_FVC_val.items():\n        fvc_true = train_csv.FVC.values[train_csv.Patient == p]\n        weeks_true = train_csv.Weeks.values[train_csv.Patient == p]\n        pente = preds_dict['pente']\n        fvc_pred = pente * (weeks_true[0] - weeks_true) + fvc_true[0]\n        fvc_true_list.append(fvc_true)\n        fvc_pred_list.append(fvc_pred)\n \n    best_sigma, best_score_val = calibrate_best_sigma(fvc_true_list, fvc_pred_list)\n    print(f\"Sigma calibré : {best_sigma:.1f}  |  Score validation (modèle image) : {best_score_val:.4f}\")\n \n    test_csv = pd.read_csv(f'{ROOT}/test.csv')\n    P_FVC_test = predict_fvc_from_images(\n        model, test_csv.Patient.unique(), test_csv, f'{ROOT}/test', is_train=False\n    )\n \n    sub_img = pd.read_csv(f'{ROOT}/sample_submission.csv')\n    for k in sub_img.Patient_Week.values:\n        p, w = k.split('_')\n        w = int(w)\n        if p not in P_FVC_test:\n            continue\n \n        fvc_init  = test_csv.FVC.values[test_csv.Patient == p][0]\n        week_init = test_csv.Weeks.values[test_csv.Patient == p][0]\n \n        pente = P_FVC_test[p]['pente']\n \n        sub_img.loc[sub_img.Patient_Week == k, 'FVC'] = pente * (w - week_init) + fvc_init\n        sub_img.loc[sub_img.Patient_Week == k, 'Confidence'] = best_sigma\n \n    subs.append(sub_img[[\"Patient_Week\", \"FVC\", \"Confidence\"]].copy())\n \nN = len(subs)\nimg_sub = subs[0].copy()\nimg_sub[\"FVC\"] = sum(s[\"FVC\"] for s in subs) / N\nimg_sub[\"Confidence\"] = sum(s[\"Confidence\"] for s in subs) / N\n \nimg_sub[\"FVC\"] = np.clip(img_sub[\"FVC\"], 100, 6000)\nimg_sub[\"Confidence\"] = np.clip(img_sub[\"Confidence\"], 70, 1000)\n \nimg_sub.to_csv(\"submission_img.csv\", index=False)\nprint(\"Fichier submission_img.csv généré avec succès !\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:38:09.986217Z","iopub.execute_input":"2026-08-04T14:38:09.986671Z","iopub.status.idle":"2026-08-04T14:44:46.699837Z","shell.execute_reply.started":"2026-08-04T14:38:09.986630Z","shell.execute_reply":"2026-08-04T14:44:46.699005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 9. PRÉPARATION DES DONNÉES TABULAIRES (modèle de régression)\n# ─────────────────────────────────────────────────────────────────────────────\ntr_tab  = pd.read_csv(f\"{ROOT}/train.csv\")\ntr_tab.drop_duplicates(keep=False, inplace=True, subset=['Patient', 'Weeks'])\nchunk   = pd.read_csv(f\"{ROOT}/test.csv\")\nsub_raw = pd.read_csv(f\"{ROOT}/sample_submission.csv\")\n \nsub_raw['Patient'] = sub_raw['Patient_Week'].apply(lambda x: x.split('_')[0])\nsub_raw['Weeks']   = sub_raw['Patient_Week'].apply(lambda x: int(x.split('_')[-1]))\nsub_raw = sub_raw[['Patient', 'Weeks', 'Confidence', 'Patient_Week']]\nsub_raw = sub_raw.merge(chunk.drop('Weeks', axis=1), on=\"Patient\")\n \ntr_tab['WHERE']  = 'train'\nchunk['WHERE']   = 'val'\nsub_raw['WHERE'] = 'test'\ndata = pd.concat([tr_tab, chunk, sub_raw], ignore_index=True)\nprint(f\"Tailles — train:{tr_tab.shape}, val:{chunk.shape}, test:{sub_raw.shape}, total:{data.shape}\")\n \ndata['min_week'] = data['Weeks']\ndata.loc[data.WHERE == 'test', 'min_week'] = np.nan\ndata['min_week'] = data.groupby('Patient')['min_week'].transform('min')\n \nbase = data.loc[data.Weeks == data.min_week, ['Patient', 'FVC']].copy() \nbase.columns = ['Patient', 'min_FVC']\nbase['nb'] = base.groupby('Patient').cumcount() + 1\nbase = base[base.nb == 1].drop('nb', axis=1)\ndata = data.merge(base, on='Patient', how='left')\ndata['base_week'] = data['Weeks'] - data['min_week']\n \nCOLS = ['Sex', 'SmokingStatus']\nFE = []\nfor col in COLS:\n    for mod in data[col].unique():\n        FE.append(mod)\n        data[mod] = (data[col] == mod).astype(int)\n \nfor col_raw, col_new in [('Age', 'age'), ('min_FVC', 'BASE'),\n                          ('base_week', 'week'), ('Percent', 'percent')]:\n    mn, mx = data[col_raw].min(), data[col_raw].max()\n    data[col_new] = (data[col_raw] - mn) / (mx - mn)\n    FE.append(col_new)\n \ntr_tab  = data.loc[data.WHERE == 'train']\nchunk   = data.loc[data.WHERE == 'val']\nsub_raw = data.loc[data.WHERE == 'test']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:44:46.701866Z","iopub.execute_input":"2026-08-04T14:44:46.702324Z","iopub.status.idle":"2026-08-04T14:44:46.756639Z","shell.execute_reply.started":"2026-08-04T14:44:46.702301Z","shell.execute_reply":"2026-08-04T14:44:46.755764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 10. MODÈLE TABULAIRE (Laplace Loss, deux sorties : FVC + Sigma)\n# ─────────────────────────────────────────────────────────────────────────────\nimport os\nimport time\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport tensorflow.keras.backend as K\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.models as M\nfrom sklearn.model_selection import StratifiedGroupKFold\n\n# --- Fixation des seeds pour la reproductibilité ---\ndef seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything(42)\n\n# --- Constantes pour la perte de Laplace ---\nC1 = tf.constant(80, dtype='float32')\nC2 = tf.constant(1000, dtype='float32')\n\ndef laplace_loss_train(y_true, y_pred):\n    y_true     = tf.cast(y_true, tf.float32)\n    fvc_pred   = y_pred[:, 0:1]\n    sigma      = y_pred[:, 1:2]\n    sigma_clip = tf.maximum(sigma, C1)\n    delta      = tf.abs(y_true - fvc_pred)\n    sq2        = tf.sqrt(tf.constant(2, dtype=tf.float32))\n    metric     = -(delta / sigma_clip) * sq2 - tf.math.log(sigma_clip * sq2)\n    return -K.mean(metric)\n\ndef osic_competition_metric(y_true, y_pred):\n    y_true     = tf.cast(y_true, tf.float32)\n    fvc_pred   = y_pred[:, 0:1]\n    sigma      = y_pred[:, 1:2]\n    sigma_clip = tf.maximum(sigma, C1)\n    delta      = tf.minimum(tf.abs(y_true - fvc_pred), C2)\n    sq2        = tf.sqrt(tf.constant(2, dtype=tf.float32))\n    metric     = -(delta / sigma_clip) * sq2 - tf.math.log(sigma_clip * sq2)\n    return -K.mean(metric)\n\ndef make_model(nh, lr=0.0005):\n    z = L.Input((nh,), name=\"Patient\")\n    x = L.Dense(100, activation=\"relu\", name=\"d1\")(z)\n    x = L.Dense(100, activation=\"relu\", name=\"d2\")(x)\n    out_fvc   = L.Dense(1, name=\"fvc_output\")(x)\n    out_sigma = L.Dense(1, activation=\"relu\", name=\"sigma_output\")(x)\n    preds     = L.Concatenate(name=\"preds\")([out_fvc, out_sigma])\n    \n    model = M.Model(z, preds, name=\"FVC_Sigma_Model\")\n    model.compile(\n        loss=laplace_loss_train,\n        metrics=[osic_competition_metric],\n        optimizer=tf.keras.optimizers.Adam(learning_rate=lr)\n    )\n    return model\n\n# --- Préparation des données tabulaires ---\ny  = tr_tab['FVC'].values\nz  = tr_tab[FE].values\nze = sub_raw[FE].values\nnh = z.shape[1]\npatient_ids = tr_tab['Patient'].values\n\n# ─────────────────────────────────────────────────────────────────────────────\n# 10.1 RECHERCHE DU LEARNING RATE (LR Finder)\n# ─────────────────────────────────────────────────────────────────────────────\nclass LRFinder(tf.keras.callbacks.Callback):\n    def __init__(self, min_lr=1e-5, max_lr=1e-1, steps=100):\n        super().__init__()\n        self.min_lr = min_lr\n        self.max_lr = max_lr\n        self.steps = steps\n        self.lrs, self.losses = [], []\n\n    def on_train_begin(self, logs=None):\n        self.lr_schedule = np.logspace(\n            np.log10(self.min_lr), \n            np.log10(self.max_lr), \n            num=self.steps\n        )\n        self.current_step = 0\n\n    def on_batch_end(self, batch, logs=None):\n        if self.current_step < self.steps:\n            lr = self.lr_schedule[self.current_step]\n            try:\n                self.model.optimizer.learning_rate.assign(lr)\n            except AttributeError:\n                tf.keras.backend.set_value(self.model.optimizer.lr, lr)\n                \n            self.lrs.append(lr)\n            self.losses.append(logs.get('loss'))\n            self.current_step += 1\n        else:\n            self.model.stop_training = True\n\nprint(\"\\n--- Lancement du LR Finder ---\")\ndummy_model = make_model(nh)\n# Petite taille de batch (16) pour garantir au moins 100 étapes d'apprentissage\nlr_finder = LRFinder(min_lr=1e-5, max_lr=1e-1, steps=100)\ndummy_model.fit(z, y, batch_size=16, epochs=1, callbacks=[lr_finder], verbose=0)\n\nplt.figure(figsize=(8, 4))\nplt.plot(lr_finder.lrs, lr_finder.losses)\nplt.xscale('log')\nplt.xlabel('Learning Rate')\nplt.ylabel('Loss')\nplt.title('LR Finder - Modèle Tabulaire')\nplt.grid(True)\nplt.show()\n\n# ─────────────────────────────────────────────────────────────────────────────\n# 10.2 ENTRAÎNEMENT STRATIFIED GROUP K-FOLD (Complet avec Métriques & Graphes)\n# ─────────────────────────────────────────────────────────────────────────────\nstart = time.time()\n\n# Hyperparamètres recommandés\nBEST_LR        = 1e-2\nEPOCHS         = 15\nBATCH_SIZE_TAB = 128\nn_splits       = 10\n\n# Stratification par quantiles de FVC\ny_bins = np.digitize(y, bins=np.percentile(y, [20, 40, 60, 80]))\n\nkf = StratifiedGroupKFold(n_splits=n_splits, shuffle=True, random_state=42)\nn_folds = kf.get_n_splits(groups=patient_ids)\n\npe   = np.zeros((ze.shape[0], 2)) # Prédictions Test\npred = np.zeros((z.shape[0], 2))  # Prédictions OOF\n\ntrain_losses, val_losses = [], []\nall_train_losses, all_val_losses = [], []\nall_train_metric, all_val_metric = [], []\n\ncnt = 0\nfor tr_idx, val_idx in kf.split(z, y_bins, groups=patient_ids):\n    cnt += 1\n    print(f\"\\n── FOLD {cnt}/{n_folds} ──\")\n\n    net = make_model(nh, lr=BEST_LR)\n\n    checkpoint_path = f\"meilleur_modele_fold_{cnt}.weights.h5\"\n    checkpoint = tf.keras.callbacks.ModelCheckpoint(\n        filepath=checkpoint_path,\n        monitor=\"val_loss\",\n        save_best_only=True,\n        mode=\"min\",\n        save_weights_only=True\n    )\n\n    history = net.fit(\n        z[tr_idx], y[tr_idx],\n        batch_size=BATCH_SIZE_TAB,\n        epochs=EPOCHS,\n        validation_data=(z[val_idx], y[val_idx]),\n        callbacks=[checkpoint],\n        verbose=0\n    )\n\n    net.load_weights(checkpoint_path)\n\n    # Évaluation\n    train_loss, train_metric = net.evaluate(z[tr_idx], y[tr_idx], verbose=0, batch_size=BATCH_SIZE_TAB)\n    val_loss, val_metric     = net.evaluate(z[val_idx], y[val_idx], verbose=0, batch_size=BATCH_SIZE_TAB)\n    \n    train_losses.append(train_loss)\n    val_losses.append(val_loss)\n\n    print(f\"  train loss : {train_loss:.4f}  |  val loss : {val_loss:.4f}  |  train metric : {train_metric:.4f}  |  val metric : {val_metric:.4f}\")\n\n    all_train_losses.append(history.history['loss'])\n    all_val_losses.append(history.history['val_loss'])\n    all_train_metric.append(history.history['osic_competition_metric'])       \n    all_val_metric.append(history.history['val_osic_competition_metric'])     \n\n    # Prédictions\n    pred[val_idx] = net.predict(z[val_idx], batch_size=BATCH_SIZE_TAB, verbose=0)\n    pe          += net.predict(ze,         batch_size=BATCH_SIZE_TAB, verbose=0) / n_folds\n\n    # Nettoyage du fichier temporaire de poids\n    if os.path.exists(checkpoint_path):\n        os.remove(checkpoint_path)\n\n# --- Affichage des résultats globaux ---\nprint(f\"\\n✓ Val loss moyenne   : {np.mean(val_losses):.4f} ± {np.std(val_losses):.4f}\")\nprint(f\"✓ Train loss moyenne : {np.mean(train_losses):.4f} ± {np.std(train_losses):.4f}\")\n\n# --- Courbes moyennes de Loss et Métrique ---\nplt.figure(figsize=(10, 5))\nplt.plot(np.mean(all_train_losses, axis=0), label='Moyenne Train Loss', color='blue', lw=2)\nplt.plot(np.mean(all_val_losses, axis=0),   label='Moyenne Val Loss',   color='red',  lw=2)\nplt.title(\"Courbe d'apprentissage — Moyenne des Folds (Loss)\")\nplt.xlabel('Époques')\nplt.ylabel('Laplace Loss')\nplt.legend()\nplt.grid(True)\nplt.show()\n\n# --- Courbes par Fold ---\nfig, axes = plt.subplots(n_folds, 2, figsize=(12, 3 * n_folds))\nfor i in range(n_folds):\n    # Périmètre Loss\n    axes[i, 0].plot(all_train_losses[i], label='Train Loss', color='blue', lw=1.5)\n    axes[i, 0].plot(all_val_losses[i],   label='Val Loss',   color='red',  lw=1.5)\n    axes[i, 0].set_title(f'Fold {i+1} — Laplace Loss')\n    axes[i, 0].grid(True, alpha=0.3)\n    axes[i, 0].legend(fontsize=8)\n\n    # Périmètre Métrique\n    axes[i, 1].plot(-np.array(all_train_metric[i]), label='Train Metric', color='green',  lw=1.5)\n    axes[i, 1].plot(-np.array(all_val_metric[i]),   label='Val Metric',   color='orange', lw=1.5)\n    axes[i, 1].set_title(f'Fold {i+1} — Score OSIC')\n    axes[i, 1].grid(True, alpha=0.3)\n    axes[i, 1].legend(fontsize=8)\n\nplt.suptitle(\"Loss et Métrique de compétition — par Fold\", fontsize=14)\nplt.tight_layout()\nplt.show()\n\nelapsed = time.time() - start\nprint(f\"✓ Entraînement terminé en {elapsed//60:.0f} min {elapsed%60:.0f} sec\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:44:46.757960Z","iopub.execute_input":"2026-08-04T14:44:46.758728Z","iopub.status.idle":"2026-08-04T14:46:00.988544Z","shell.execute_reply.started":"2026-08-04T14:44:46.758679Z","shell.execute_reply":"2026-08-04T14:46:00.987770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 12. ÉVALUATION DU MODÈLE TABULAIRE\n# ─────────────────────────────────────────────────────────────────────────────\nmae_fvc    = mean_absolute_error(y, pred[:, 0])\nsigma_vals = pred[:, 1]\nsigma_mean = np.mean(sigma_vals)\n \nprint(f\"\\nErreur absolue moyenne sur FVC (MAE) : {mae_fvc:.2f}\")\nprint(f\"Confiance moyenne (Sigma)            : {sigma_mean:.2f}\")\n \nplt.hist(sigma_vals, bins=30)\nplt.title(\"Distribution de l'incertitude (Sigma) — prédictions out-of-fold\")\nplt.xlabel(\"Sigma\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:46:00.989572Z","iopub.execute_input":"2026-08-04T14:46:00.989886Z","iopub.status.idle":"2026-08-04T14:46:01.136210Z","shell.execute_reply.started":"2026-08-04T14:46:00.989855Z","shell.execute_reply":"2026-08-04T14:46:01.135364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#─────────────────────────────────────────────────────────────────────────────\n# 13. GÉNÉRATION DE LA SOUMISSION TABULAIRE\n# ─────────────────────────────────────────────────────────────────────────────\nsub_raw = sub_raw.copy()\nsub_raw['FVC1']        = pe[:, 0]\nsub_raw['Confidence1'] = pe[:, 1]\n \nsubm = sub_raw[['Patient_Week', 'FVC', 'Confidence', 'FVC1', 'Confidence1']].copy()\nprint('subm')\nsubm.loc[~subm.FVC1.isnull(), 'FVC'] = subm.loc[~subm.FVC1.isnull(), 'FVC1']\n \nif sigma_mean < 75:\n    subm['Confidence'] = mae_fvc\nelse:\n    subm.loc[~subm.FVC1.isnull(), 'Confidence'] = subm.loc[~subm.FVC1.isnull(), 'Confidence1']\n \notest = pd.read_csv(f'{ROOT}/test.csv')\nfor i in range(len(otest)):\n    key = f\"{otest.Patient[i]}_{otest.Weeks[i]}\"\n    subm.loc[subm['Patient_Week'] == key, 'FVC']        = otest.FVC[i]\n    subm.loc[subm['Patient_Week'] == key, 'Confidence'] = 0.1\n \nreg_sub = subm[['Patient_Week', 'FVC', 'Confidence']].copy()\nreg_sub.to_csv(\"submission_reg.csv\", index=False)\nprint(subm.describe().T)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:46:01.137443Z","iopub.execute_input":"2026-08-04T14:46:01.137940Z","iopub.status.idle":"2026-08-04T14:46:01.183597Z","shell.execute_reply.started":"2026-08-04T14:46:01.137899Z","shell.execute_reply":"2026-08-04T14:46:01.182875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────────────────────────────────────────────────────\n# 14. BLENDING FINAL\n# ─────────────────────────────────────────────────────────────────────────────\nassert len(img_sub) == len(reg_sub), (\n    f\"Tailles différentes : img_sub={len(img_sub)}, reg_sub={len(reg_sub)}\"\n)\nmissing_in_reg = set(img_sub.Patient_Week) - set(reg_sub.Patient_Week)\nmissing_in_img = set(reg_sub.Patient_Week) - set(img_sub.Patient_Week)\nassert not missing_in_reg and not missing_in_img, (\n    f\"Patient_Week non alignés : manquants dans reg={missing_in_reg}, \"\n    f\"manquants dans img={missing_in_img}\"\n)\n \nfinal = img_sub.merge(\n    reg_sub,\n    on='Patient_Week',\n    suffixes=('_img', '_reg')\n)\n \nW_IMG, W_REG = 1.0, 0.0\nfinal['FVC'] = W_IMG * final['FVC_img'] + W_REG * final['FVC_reg']\nfinal['Confidence'] = W_IMG * final['Confidence_img'] + W_REG * final['Confidence_reg']\n \nfinal['Confidence'] = np.clip(final['Confidence'], 70, 1000)\nfinal['FVC'] = np.clip(final['FVC'], 100, 6000)\n \nfinal = final[['Patient_Week', 'FVC', 'Confidence']]\n \nprint(final.head())\nprint(final.describe().T)\n \nfinal.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv généré avec succès, alignement vérifié.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-04T14:46:01.184619Z","iopub.execute_input":"2026-08-04T14:46:01.184934Z","iopub.status.idle":"2026-08-04T14:46:01.212382Z","shell.execute_reply.started":"2026-08-04T14:46:01.184900Z","shell.execute_reply":"2026-08-04T14:46:01.211605Z"}},"outputs":[],"execution_count":null}]}