{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14456136,"sourceType":"competition"},{"sourceId":13746211,"sourceType":"datasetVersion","datasetId":8746873},{"sourceId":13747057,"sourceType":"datasetVersion","datasetId":8747501},{"sourceId":13747757,"sourceType":"datasetVersion","datasetId":8747969},{"sourceId":13747854,"sourceType":"datasetVersion","datasetId":8748029},{"sourceId":13751233,"sourceType":"datasetVersion","datasetId":8750011}],"dockerImageVersionId":31192,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # CODE À EXÉCUTER DANS VOTRE NOUVEAU NOTEBOOK (avec l'indexation complète)\n# import os\n# import pandas as pd\n# import numpy as np\n# import matplotlib.pyplot as plt\n# from sklearn.model_selection import train_test_split\n# # Ajoutez cv2 si vous prévoyez de l'utiliser pour la lecture/redimensionnement\n# # import cv2 \n\n# # --- 0. Définition des chemins de base ---\n# # DATA_DIR = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/\"\n# # TRAIN_IMAGES_DIR = os.path.join(DATA_DIR, \"train_images\")\n# # TRAIN_MASKS_DIR = os.path.join(DATA_DIR, \"train_masks\")\n# # SUPPLEMENTAL_IMAGES_DIR = os.path.join(DATA_DIR, \"supplemental_images\")\n# # SUPPLEMENTAL_MASKS_DIR = os.path.join(DATA_DIR, \"supplemental_masks\")\n\n# # Fonction de recherche des masques (format corrigé : [case_id].npy)\n# # def find_mask_paths(case_id, all_mask_files, mask_dir_list):\n# #     \"\"\"\n# #     Cherche le masque NPY unique au format '[case_id].npy' en vérifiant tous les répertoires de masques.\n# #     \"\"\"\n# #     mask_file_name = str(case_id) + '.npy'\n    \n# #     # 1. Parcourir la liste combinée des fichiers de masques\n# #     if mask_file_name in all_mask_files:\n        \n# #         # 2. Trouver dans quel répertoire il se trouve\n# #         for mask_dir in mask_dir_list:\n# #             full_path = os.path.join(mask_dir, mask_file_name)\n# #             if os.path.exists(full_path):\n# #                 return [full_path]\n# #     return []\n\n# # --- 1. Lister tous les fichiers PNG d'entraînement (Inclus les supplémentaires) ---\n# # all_train_images = []\n# # image_dirs = [\n# #     os.path.join(TRAIN_IMAGES_DIR, 'authentic'),\n# #     os.path.join(TRAIN_IMAGES_DIR, 'forged'),\n# #     SUPPLEMENTAL_IMAGES_DIR \n# # ]\n\n# # print(\"--- Indexation des Images ---\")\n# # for img_dir in image_dirs:\n# #     if os.path.exists(img_dir):\n# #         is_forged = ('forged' in img_dir)\n        \n# #         for img_name in os.listdir(img_dir):\n# #             if img_name.endswith('.png'):\n# #                 case_id = img_name.replace('.png', '')\n                \n# #                 # Pour les supplemental, on ne peut pas savoir si c'est forged sans le masque\n# #                 is_forged_final = is_forged if 'train_images' in img_dir else False \n                \n# #                 all_train_images.append({'case_id': case_id, \n# #                                          'path': os.path.join(img_dir, img_name),\n# #                                          'is_forged': is_forged_final})\n\n# # # --- 2. Créer le DataFrame et gérer les doublons (Priorité au cas 'forged') ---\n# # train_df = pd.DataFrame(all_train_images)\n\n# # # Trier pour mettre les cas Forged en dernier pour que drop_duplicates les conserve\n# # train_df = train_df.sort_values(by='is_forged').reset_index(drop=True)\n# # train_df = train_df.drop_duplicates(subset=['case_id'], keep='last').reset_index(drop=True)\n\n# # # --- 3. Intégrer les chemins complets des masques ---\n\n# # mask_dir_list = [TRAIN_MASKS_DIR, SUPPLEMENTAL_MASKS_DIR]\n# # all_mask_files = []\n# # for mask_dir in mask_dir_list:\n# #     if os.path.exists(mask_dir):\n# #         all_mask_files.extend(os.listdir(mask_dir))\n\n# # print(f\"\\nTotal des fichiers de masques (train + supplemental) trouvés : {len(all_mask_files)}\")\n\n# # # Appliquer la fonction pour ajouter les chemins des masques au DataFrame\n# # train_df['mask_paths'] = train_df['case_id'].astype(str).apply(\n# #     lambda x: find_mask_paths(x, all_mask_files, mask_dir_list))\n# # train_df['mask_count'] = train_df['mask_paths'].apply(len)\n\n# # # CORRECTION: Mettre à jour 'is_forged' pour toutes les images ayant un masque\n# # train_df.loc[train_df['mask_count'] > 0, 'is_forged'] = True\n\n# # # Filtrer les images incohérentes (déclarées forged mais sans masque)\n# # train_df_clean = train_df[~((train_df['is_forged'] == True) & (train_df['mask_count'] == 0))].reset_index(drop=True)\n\n# # print(f\"\\nTotal des images d'entraînement uniques et utilisables : {len(train_df_clean)}\")\n# # print(f\"Images avec masques (Forged) : {train_df_clean['is_forged'].sum()}\")\n# # print(f\"Images sans masques (Authentic) : {len(train_df_clean) - train_df_clean['is_forged'].sum()}\")\n\n\n# # # --- 4. Séparation Train/Validation (Stratifiée) ---\n# # VALID_SIZE = 0.20\n# # SEED = 42 \n\n# # # S'assurer que la stratification fonctionne sur 'is_forged'\n# # train_ids, valid_ids = train_test_split(\n# #     train_df_clean['case_id'].values,\n# #     test_size=VALID_SIZE,\n# #     random_state=SEED,\n# #     # La stratification est essentielle car les classes Forged/Authentic sont déséquilibrées\n# #     stratify=train_df_clean['is_forged'].values\n# # )\n\n# # train_df_final = train_df_clean[train_df_clean['case_id'].isin(train_ids)].reset_index(drop=True)\n# # valid_df_final = train_df_clean[train_df_clean['case_id'].isin(valid_ids)].reset_index(drop=True)\n\n# # print(\"\\n--- Répartition des Jeux de Données ---\")\n# # print(f\"Jeu d'entraînement (Train) : {len(train_df_final)} images\")\n# # print(f\"Jeu de validation (Validation) : {len(valid_df_final)} images\")\n# # print(f\"Proportion Forged/Total dans le Train : {train_df_final['is_forged'].sum() / len(train_df_final):.3f}\")\n# # print(f\"Proportion Forged/Total dans la Validation : {valid_df_final['is_forged'].sum() / len(valid_df_final):.3f}\")\n\n# # # --- Assurez-vous d'avoir les DataFrames de travail ---\n# # print(\"\\nLes DataFrames train_df_final et valid_df_final sont prêts à être utilisés.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:08:09.787203Z","iopub.execute_input":"2025-11-14T21:08:09.787501Z","iopub.status.idle":"2025-11-14T21:08:17.341593Z","shell.execute_reply.started":"2025-11-14T21:08:09.787466Z","shell.execute_reply":"2025-11-14T21:08:17.340537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import os\n# # import numpy as np\n# # import warnings\n\n# # # --- Définition des Paramètres Globaux Manquants ---\n# # IMG_SIZE = 256 \n# # # ----------------------------------------------------\n\n# # # Supprimer les avertissements qui peuvent être générés par np.load()\n# # warnings.filterwarnings(\"ignore\", category=UserWarning)\n\n# # # --- Fonction de Vérification du Masque ---\n# # def is_mask_corrupted(mask_path, img_size):\n# #     \"\"\"Vérifie si le fichier masque est chargeable et a des dimensions valides.\"\"\"\n# #     try:\n# #         mask_data = np.load(mask_path)\n        \n# #         # Logique de combinaison des canaux (comme dans le générateur)\n# #         if mask_data.ndim == 3 and mask_data.shape[0] in [2, 3]:\n# #             mask_2d = np.max(mask_data, axis=0)\n# #         else:\n# #             mask_2d = mask_data\n            \n# #         # Vérification de la taille et de la forme\n# #         if not (mask_2d.size > 0 and mask_2d.ndim >= 2 and mask_2d.shape[0] > 0 and mask_2d.shape[1] > 0):\n# #             return True # Masque vide ou de forme non valide\n\n# #         return False\n# #     except Exception:\n# #         return True # Le fichier ne peut pas être chargé\n\n# # # --- Nettoyage du DataFrame ---\n\n# # print(f\"Début de la vérification de {len(train_df_final)} échantillons...\")\n# # rows_to_drop = []\n\n# # # Itérer sur une copie pour éviter les problèmes d'itération lors de la suppression\n# # for index, row in train_df_final.iterrows():\n# #     # Nous vérifions le premier masque seulement, car il est le seul utilisé\n# #     mask_path = row['mask_paths'][0] \n    \n# #     if is_mask_corrupted(mask_path, IMG_SIZE):\n# #         rows_to_drop.append(index)\n        \n# # if rows_to_drop:\n# #     # Suppression des lignes corrompues\n# #     train_df_final = train_df_final.drop(rows_to_drop).reset_index(drop=True)\n# #     print(f\"\\n✅ Nettoyage terminé. {len(rows_to_drop)} échantillons corrompus ont été retirés.\")\n# # else:\n# #     print(\"\\n✅ Nettoyage terminé. Aucune ligne corrompue trouvée.\")\n\n# # print(f\"Taille finale du jeu d'entraînement : {len(train_df_final)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:10:36.735833Z","iopub.execute_input":"2025-11-14T21:10:36.736807Z","iopub.status.idle":"2025-11-14T21:10:59.207544Z","shell.execute_reply.started":"2025-11-14T21:10:36.736691Z","shell.execute_reply":"2025-11-14T21:10:59.206327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- INSTALLATION NÉCESSAIRE (Si ce n'est pas déjà fait) ---\n# # !pip install albumentations\n# # !pip install opencv-python\n\n# # from tensorflow.keras.utils import Sequence\n# # import cv2 \n# # import numpy as np\n# # import albumentations as A\n\n# # # --- Paramètres ---\n# # IMG_SIZE = 256 \n# # BATCH_SIZE = 8 \n# # # ---\n\n# # # --- DÉFINITION DE LA PIPELINE D'AUGMENTATION ---\n# # def get_training_augmentation(img_size):\n# #     \"\"\"Pipeline d'augmentation pour le jeu d'entraînement.\"\"\"\n# #     train_transform = A.Compose([\n# #         A.HorizontalFlip(p=0.5),\n# #         A.VerticalFlip(p=0.5),\n# #         A.ShiftScaleRotate(scale_limit=0.1, rotate_limit=10, shift_limit=0.1, p=0.5, \n# #                             border_mode=cv2.BORDER_CONSTANT),\n# #         A.RandomBrightnessContrast(brightness_limit=0.1, contrast_limit=0.1, p=0.2),\n# #         A.GaussNoise(p=0.2), \n# #     ], p=1.0)\n# #     return train_transform\n\n# # # --- CLASSE DE GÉNÉRATEUR AVEC CORRECTION D'ERREURS ---\n# # class ScientificImageGenerator(Sequence):\n# #     def __init__(self, df, batch_size=BATCH_SIZE, img_size=IMG_SIZE, shuffle=True, augment=False):\n# #         self.df = df\n# #         self.batch_size = batch_size\n# #         self.img_size = img_size\n# #         self.shuffle = shuffle\n# #         self.augment = augment\n# #         self.on_epoch_end()\n        \n# #         if self.augment:\n# #             self.transform = get_training_augmentation(self.img_size)\n# #         else:\n# #             self.transform = A.Compose([]) \n\n# #     def __len__(self):\n# #         return int(np.floor(len(self.df) / self.batch_size))\n\n# #     def on_epoch_end(self):\n# #         self.indices = np.arange(len(self.df))\n# #         if self.shuffle == True:\n# #             np.random.shuffle(self.indices)\n\n# #     def __getitem__(self, index):\n# #         indices = self.indices[index * self.batch_size:(index + 1) * self.batch_size]\n# #         batch_df = self.df.iloc[indices]\n        \n# #         X = np.empty((self.batch_size, self.img_size, self.img_size, 3), dtype=np.float32)\n# #         y = np.empty((self.batch_size, self.img_size, self.img_size, 1), dtype=np.float32)\n\n# #         for i_local, row in enumerate(batch_df.itertuples()):\n# #             is_corrupted = False\n            \n# #             # --- 1. Chargement de l'Image ---\n# #             img = cv2.imread(row.path)\n# #             if img is None:\n# #                 is_corrupted = True\n                \n# #             if not is_corrupted:\n# #                 img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n# #             # --- 2. Chargement et Pré-traitement du Masque ---\n# #             if not is_corrupted:\n# #                 try:\n# #                     mask_data = np.load(row.mask_paths[0]) \n# #                 except Exception:\n# #                     is_corrupted = True # Le chargement du .npy a échoué\n                    \n# #             if not is_corrupted:\n# #                 # COMBINAISON DES CANAUX (X, H, W) -> (H, W)\n# #                 if mask_data.ndim == 3 and mask_data.shape[0] in [2, 3]:\n# #                     mask_2d = np.max(mask_data, axis=0)\n# #                 else:\n# #                     mask_2d = mask_data\n                    \n# #                 # VÉRIFICATION DE VALIDITÉ ULTRA-SÛRE\n# #                 is_mask_valid = (mask_2d.size > 0 and mask_2d.ndim >= 2 and mask_2d.shape[0] > 0 and mask_2d.shape[1] > 0)\n                \n# #                 if not is_mask_valid:\n# #                     is_corrupted = True\n# #                 else:\n# #                     mask_2d_uint8 = mask_2d.astype(np.uint8) \n\n# #             # ⭐️ BLOC TRY/EXCEPT POUR CONTOURNER L'ERREUR CV2 ⭐️\n# #             try:\n# #                 if is_corrupted:\n# #                     # Remplir avec des zéros si le fichier est invalide\n# #                     img_initial_resize = np.zeros((self.img_size, self.img_size, 3), dtype=np.uint8)\n# #                     mask_initial_resize = np.zeros((self.img_size, self.img_size), dtype=np.uint8)\n# #                 else:\n# #                     # Tentative de redimensionnement (point de crash potentiel)\n# #                     img_initial_resize = cv2.resize(img, (self.img_size, self.img_size))\n# #                     mask_initial_resize = cv2.resize(mask_2d_uint8, (self.img_size, self.img_size), \n# #                                                      interpolation=cv2.INTER_NEAREST) \n# #             except cv2.error as e:\n# #                 # Si cv2 crash, forcer les zéros et ignorer ce fichier\n# #                 print(f\"FATAL CV2 ERROR: Redimensionnement échoué pour l'échantillon {row.case_id}. Remplacement par des zéros.\")\n# #                 img_initial_resize = np.zeros((self.img_size, self.img_size, 3), dtype=np.uint8)\n# #                 mask_initial_resize = np.zeros((self.img_size, self.img_size), dtype=np.uint8)\n\n# #             # --- 3. APPLICATION DE L'AUGMENTATION ---\n# #             if img_initial_resize.ndim == 2:\n# #                  img_initial_resize = cv2.cvtColor(img_initial_resize, cv2.COLOR_GRAY2RGB)\n\n# #             augmented = self.transform(image=img_initial_resize, mask=mask_initial_resize)\n# #             img_aug = augmented['image']\n# #             mask_aug = augmented['mask']\n            \n# #             # ⭐️ CORRECTION FINALE : FORCER LE MASQUE À ÊTRE 2D ⭐️\n# #             try:\n# #                 # Remodeler le masque pour garantir qu'il soit (IMG_SIZE, IMG_SIZE)\n# #                 mask_reshaped = mask_aug.reshape(self.img_size, self.img_size)\n# #             except ValueError:\n# #                 # Si le remodelage échoue, le masque est malformé par Albumentations. Remplacer par un masque vide.\n# #                 print(f\"ATTENTION: Masque de sortie malformé pour l'échantillon {row.case_id}. Remplacé par des zéros.\")\n# #                 mask_reshaped = np.zeros((self.img_size, self.img_size), dtype=np.uint8)\n\n\n# #             # Normalisation et stockage\n# #             X[i_local] = img_aug / 255.0 \n            \n# #             # Ajouter la dimension du canal pour le masque (H, W) -> (H, W, 1)\n# #             y[i_local] = np.expand_dims(mask_reshaped > 0, axis=-1)\n            \n# #         return X, y\n\n# # # --- Initialisation des Générateurs ---\n# # # (Ces lignes supposent que train_df_final et valid_df_final sont définis)\n# # # train_gen = ScientificImageGenerator(train_df_final, batch_size=BATCH_SIZE, shuffle=True, augment=True)\n# # # valid_gen = ScientificImageGenerator(valid_df_final, batch_size=BATCH_SIZE, shuffle=False, augment=False) \n\n# # # print(f\"Générateur d'entraînement (avec augmentation) : Lots par époque = {len(train_gen)}\")\n# # # print(f\"Générateur de validation (sans augmentation) : Lots par époque = {len(valid_gen)}\")\n\n# # # --- Vérification d'un Lot ---\n# # # X_sample, y_sample = train_gen[0]\n# # # print(f\"\\nForme du Lot d'Images (X) : {X_sample.shape}\")\n# # # print(f\"Forme du Lot de Masques (y) : {y_sample.shape}\")\n# # # print(f\"Vérification : Max/Min des masques : {np.max(y_sample)}, {np.min(y_sample)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:11:09.326808Z","iopub.execute_input":"2025-11-14T21:11:09.327615Z","iopub.status.idle":"2025-11-14T21:11:58.092790Z","shell.execute_reply.started":"2025-11-14T21:11:09.327566Z","shell.execute_reply":"2025-11-14T21:11:58.091667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # !git clone https://github.com/qubvel/segmentation_models.git\n# # !git clone https://github.com/qubvel/classification_models.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:21:38.185807Z","iopub.execute_input":"2025-11-14T21:21:38.186268Z","iopub.status.idle":"2025-11-14T21:21:38.477028Z","shell.execute_reply.started":"2025-11-14T21:21:38.186238Z","shell.execute_reply":"2025-11-14T21:21:38.475852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # !pip install efficientnet==1.1.1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:22:40.471345Z","iopub.execute_input":"2025-11-14T21:22:40.471662Z","iopub.status.idle":"2025-11-14T21:22:44.562597Z","shell.execute_reply.started":"2025-11-14T21:22:40.471632Z","shell.execute_reply":"2025-11-14T21:22:44.561463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # !ls -l ./classification_models","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:24:09.218119Z","iopub.execute_input":"2025-11-14T21:24:09.218478Z","iopub.status.idle":"2025-11-14T21:24:09.376011Z","shell.execute_reply.started":"2025-11-14T21:24:09.218447Z","shell.execute_reply":"2025-11-14T21:24:09.374875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import efficientnet\n# # print(\"Importation efficientnet réussie.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:26:55.039653Z","iopub.execute_input":"2025-11-14T21:26:55.040109Z","iopub.status.idle":"2025-11-14T21:26:55.045776Z","shell.execute_reply.started":"2025-11-14T21:26:55.040080Z","shell.execute_reply":"2025-11-14T21:26:55.044549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import os\n# Revenir au répertoire de travail original (/kaggle/working)\n# # os.chdir('/kaggle/working') \n# # print(f\"Répertoire actuel : {os.getcwd()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:29:00.266066Z","iopub.execute_input":"2025-11-14T21:29:00.266423Z","iopub.status.idle":"2025-11-14T21:29:00.272458Z","shell.execute_reply.started":"2025-11-14T21:29:00.266401Z","shell.execute_reply":"2025-11-14T21:29:00.271354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Exécuter les deux commandes ensemble via shell pour maintenir le contexte du répertoire.\n# # !cd /kaggle/working/segmentation_models/segmentation_models/backbones && sed -i 's/from classification_models.models_factory import ModelsFactory/from classification_models.classification_models.models_factory import ModelsFactory/g' backbones_factory.py","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:33:40.661917Z","iopub.execute_input":"2025-11-14T21:33:40.662283Z","iopub.status.idle":"2025-11-14T21:33:40.808682Z","shell.execute_reply.started":"2025-11-14T21:33:40.662252Z","shell.execute_reply":"2025-11-14T21:33:40.807406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import os\n# # os.chdir('/kaggle/working') \n# # print(f\"Répertoire actuel : {os.getcwd()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:34:02.018895Z","iopub.execute_input":"2025-11-14T21:34:02.019249Z","iopub.status.idle":"2025-11-14T21:34:02.025555Z","shell.execute_reply.started":"2025-11-14T21:34:02.019220Z","shell.execute_reply":"2025-11-14T21:34:02.024479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import sys\n# # import os\n# # import tensorflow as tf\n# # from tensorflow.keras.optimizers import Adam\n# # # Tenter d'importer segmentation_models. L'erreur est attendue à l'intérieur de ce module.\n# # import segmentation_models as sm \n\n# # # --- CONTOURNEMENT MANUEL DU CHEMIN (FIXE LE BUG D'IMPORTATION) ---\n\n# # # 1. Ajouter le répertoire courant (pour trouver les dossiers clonés)\n# # current_dir = os.getcwd()\n# # if current_dir not in sys.path:\n# #     sys.path.append(current_dir)\n\n# # # 2. Ajouter le sous-dossier contenant le module `classification_models`\n# # # CECI DEVRAIT ÊTRE LA SOLUTION : './classification_models/classification_models'\n# # target_path = './classification_models/classification_models'\n# # if target_path not in sys.path:\n# #     sys.path.append(target_path)\n\n# # # 3. Ajouter le sous-dossier contenant le module `segmentation_models` (si nécessaire)\n# # # Le chemin simple devrait suffire maintenant\n# # if './segmentation_models' not in sys.path:\n# #     sys.path.append('./segmentation_models')\n\n# # # -------------------------------------------------------------------------------------------------\n\n# # # --- PARAMÈTRES DU MODÈLE ---\n# # IMG_SIZE = 256 \n# # N_CLASSES = 1 \n# # BACKBONE = 'resnet34' \n\n# # # --- 1. DÉFINITION DES MÉTRIQUES ET PERTES ADAPTÉES ---\n\n# # # Coefficient de Dice (Métriques)\n# # def dice_coef(y_true, y_pred, smooth=1e-7):\n# #     y_true_f = tf.keras.backend.flatten(y_true)\n# #     y_pred_f = tf.keras.backend.flatten(y_pred)\n# #     intersection = tf.keras.backend.sum(y_true_f * y_pred_f)\n# #     return (2. * intersection + smooth) / (tf.keras.backend.sum(y_true_f) + tf.keras.backend.sum(y_pred_f) + smooth)\n\n# # # Fonction de Perte (Dice Loss)\n# # def dice_loss(y_true, y_pred):\n# #     return 1.0 - dice_coef(y_true, y_pred)\n\n# # # Fonction de Perte Combinée (BCE + Dice Loss)\n# # def combined_loss(y_true, y_pred, alpha=0.5):\n# #     bce = tf.keras.losses.BinaryCrossentropy()\n# #     return alpha * dice_loss(y_true, y_pred) + (1 - alpha) * bce(y_true, y_pred)\n\n# # print(\"Métriques et fonctions de perte personnalisées définies.\")\n\n\n# # # --- 2. CONSTRUCTION ET COMPILATION DU MODÈLE U-NET ---\n\n# # # Création du modèle U-Net\n# # model = sm.Unet(\n# #     BACKBONE, \n# #     encoder_weights='imagenet', \n# #     input_shape=(IMG_SIZE, IMG_SIZE, 3), \n# #     classes=N_CLASSES, \n# #     activation='sigmoid'        \n# # )\n\n# # # Compilation du modèle\n# # model.compile(\n# #     optimizer=Adam(learning_rate=1e-4), \n# #     loss=combined_loss,\n# #     metrics=[dice_coef, 'accuracy'] \n# # )\n\n# # print(f\"\\nModèle U-Net ({BACKBONE}) construit et compilé.\")\n# # print(f\"Forme d'entrée : {model.input_shape}\")\n# # print(f\"Nombre total de paramètres : {model.count_params()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:34:06.333320Z","iopub.execute_input":"2025-11-14T21:34:06.334432Z","iopub.status.idle":"2025-11-14T21:34:08.792995Z","shell.execute_reply.started":"2025-11-14T21:34:06.334393Z","shell.execute_reply":"2025-11-14T21:34:08.791828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import numpy as np\n# # import pandas as pd\n# # import warnings\n\n# # warnings.filterwarnings(\"ignore\", category=UserWarning)\n\n# # # --- HYPOTHÈSE : train_df_final est chargé en mémoire ---\n\n# # def check_mask_for_falsification(mask_path):\n# #     \"\"\"Vérifie si le masque contient au moins un pixel non-zéro.\"\"\"\n# #     try:\n# #         # Charger le masque (.npy)\n# #         mask_data = np.load(mask_path)\n        \n# #         # Logique de combinaison des canaux (utilisée dans le générateur)\n# #         if mask_data.ndim == 3 and mask_data.shape[0] in [2, 3]:\n# #             # Utiliser le maximum sur les canaux pour obtenir le masque 2D binaire\n# #             mask_2d = np.max(mask_data, axis=0)\n# #         else:\n# #             # Masque déjà 2D ou chargement normal\n# #             mask_2d = mask_data\n            \n# #         # Un masque est \"forgé\" si la somme de ses pixels non-zéro est supérieure à zéro.\n# #         return np.any(mask_2d > 0)\n        \n# #     except Exception as e:\n# #         return False\n\n# # print(\"Début de l'analyse des masques...\")\n\n# # # 1. Appliquer la fonction pour créer la colonne 'is_forged'\n# # # train_df_final['is_forged'] = train_df_final['mask_paths'].apply(\n# # #     lambda x: check_mask_for_falsification(x[0])\n# # # )\n\n# # # 2. Séparer le jeu d'entraînement en deux DataFrames\n# # # train_df_forged = train_df_final[train_df_final['is_forged'] == True].reset_index(drop=True)\n# # # train_df_authentic = train_df_final[train_df_final['is_forged'] == False].reset_index(drop=True)\n\n# # # print(f\"\\n✅ Analyse terminée.\")\n# # # print(f\"Total échantillons d'entraînement : {len(train_df_final)}\")\n# # # print(f\"Images FORGÉES (Phase 1) : {len(train_df_forged)}\")\n# # # print(f\"Images AUTHENTIQUES : {len(train_df_authentic)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:47:17.252416Z","iopub.execute_input":"2025-11-14T21:47:17.252839Z","iopub.status.idle":"2025-11-14T21:47:26.488410Z","shell.execute_reply.started":"2025-11-14T21:47:17.252809Z","shell.execute_reply":"2025-11-14T21:47:26.487567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import numpy as np\n# # import os\n\n# # # Récupérer le chemin du masque du premier échantillon\n# # # Attention : Assurez-vous que train_df_final est dans votre mémoire et que os.getcwd() est bien /kaggle/working\n# # # first_mask_path = train_df_final['mask_paths'].iloc[0][0]\n\n# # try:\n# #     # Charger le masque\n# #     mask_data = np.load(first_mask_path)\n    \n# #     # Appliquer la même logique de combinaison que dans le générateur\n# #     if mask_data.ndim == 3 and mask_data.shape[0] in [2, 3]:\n# #         mask_2d = np.max(mask_data, axis=0)\n# #     else:\n# #         mask_2d = mask_data\n        \n# #     print(\"--- DÉBOGAGE DU MASQUE ---\")\n# #     print(f\"Forme du Masque 2D : {mask_2d.shape}\")\n# #     print(f\"Type de données (dtype) : {mask_2d.dtype}\")\n# #     print(f\"Valeur MIN du Masque : {np.min(mask_2d)}\")\n# #     print(f\"Valeur MAX du Masque : {np.max(mask_2d)}\")\n# #     print(f\"Somme des pixels du Masque : {np.sum(mask_2d)}\")\n    \n# # except Exception as e:\n# #     print(f\"Erreur lors du chargement ou de l'analyse du masque : {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:49:05.988427Z","iopub.execute_input":"2025-11-14T21:49:05.988874Z","iopub.status.idle":"2025-11-14T21:49:06.001915Z","shell.execute_reply.started":"2025-11-14T21:49:05.988835Z","shell.execute_reply":"2025-11-14T21:49:06.000688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Phase 1 : Entraînement uniquement sur les images forgées\n# # train_df_phase1 = train_df_final.copy() \n# # Le jeu de validation (valid_df_final) est conservé tel quel.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:51:13.032785Z","iopub.execute_input":"2025-11-14T21:51:13.033124Z","iopub.status.idle":"2025-11-14T21:51:13.038282Z","shell.execute_reply.started":"2025-11-14T21:51:13.033101Z","shell.execute_reply":"2025-11-14T21:51:13.037214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import cv2\n# # import numpy as np\n# # import pandas as pd\n# # from tqdm.notebook import tqdm \n\n# # # --- DATAFRAME CIBLE : VALIDATION ---\n# # # ASSUREZ-VOUS QUE 'valid_df_final' EST LE DATAFRAME QUI CONTIENT VOS DONNÉES DE VALIDATION ORIGINALES.\n# # df_to_clean = valid_df_final # <-- MODIFICATION CLÉ : CIBLE LE JEU DE VALIDATION\n# # IMG_SIZE = 256 # Doit correspondre à la taille utilisée dans le générateur\n\n# # def find_corrupt_files(df, img_size):\n# #     \"\"\"Identifie les échantillons qui provoquent des erreurs CV2 ou de masques.\"\"\"\n# #     corrupt_indices = set()\n    \n# #     for idx, row in tqdm(df.iterrows(), total=len(df), desc=\"Scanning Validation Corrupt Files\"):\n# #         # ASSUMPTION : les chemins sont dans les colonnes 'path' et 'mask_paths'\n# #         image_path = row['path'] \n# #         mask_path = row['mask_paths'][0]\n        \n# #         # 1. Tentative de chargement de l'image\n# #         img = cv2.imread(image_path)\n# #         if img is None: \n# #             corrupt_indices.add(idx)\n# #             continue\n        \n# #         # 2. Tentative de chargement du masque\n# #         try:\n# #             mask_data = np.load(mask_path)\n# #         except Exception:\n# #             corrupt_indices.add(idx)\n# #             continue\n            \n# #         if mask_data.ndim == 3 and mask_data.shape[0] in [2, 3]:\n# #             mask_2d = np.max(mask_data, axis=0)\n# #         else:\n# #             mask_2d = mask_data\n            \n# #         if mask_2d.size == 0: \n# #             pass \n            \n# #         mask_2d_uint8 = mask_2d.astype(np.uint8) \n\n# #         # 3. Tentative de Redimensionnement (la cause des erreurs FATAL CV2)\n# #         try:\n# #             # Tester le redimensionnement de l'image\n# #             cv2.resize(img, (img_size, img_size))\n# #             # Tester le redimensionnement du masque (souvent la cause principale)\n# #             cv2.resize(mask_2d_uint8, (img_size, img_size), interpolation=cv2.INTER_NEAREST)\n# #         except cv2.error:\n# #             corrupt_indices.add(idx)\n            \n# #     return list(corrupt_indices)\n\n# # # --- Exécution du nettoyage pour la validation ---\n# # print(\"Début de l'identification des fichiers corrompus dans l'ensemble de VALIDATION...\")\n# # # corrupt_validation_list = find_corrupt_files(df_to_clean, IMG_SIZE) \n# # # print(f\"✅ Identification terminée. {len(corrupt_validation_list)} échantillons corrompus de validation trouvés.\")\n\n# # # Créer le DataFrame de validation nettoyé\n# # # valid_df_clean = df_to_clean.drop(corrupt_validation_list).reset_index(drop=True)\n\n# # # print(f\"Taille du DataFrame de validation initial : {len(df_to_clean)}\")\n# # # print(f\"Taille du DataFrame de validation nettoyé : {len(valid_df_clean)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T22:41:04.626987Z","iopub.execute_input":"2025-11-14T22:41:04.628174Z","iopub.status.idle":"2025-11-14T22:41:28.582290Z","shell.execute_reply.started":"2025-11-14T22:41:04.628143Z","shell.execute_reply":"2025-11-14T22:41:28.580989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\n# # --- ASSUREZ-VOUS QUE CES VARIABLES SONT DÉFINIES DANS VOTRE NOTEBOOK ---\n# # IMG_SIZE = 256\n# # model est le modèle U-Net compilé.\n# # BATCH_SIZE = 8\n# # EPOCHS_PHASE1 = 5\n# # MODEL_NAME_PHASE1 = 'unet_resnet34_phase1.h5' \n\n# # --- 1. GÉNÉRATEURS DE DONNÉES (CORRIGÉ : Utilisation de valid_df_clean) ---\n\n# # Générateur d'entraînement Phase 1 (sur les images forgées NETTOYÉES)\n# # train_generator_phase1 = ScientificImageGenerator(\n# #     train_df_clean, # <-- DATAFRAME NETTOYÉ (1201 échantillons)\n# #     batch_size=BATCH_SIZE, \n# #     img_size=IMG_SIZE, \n# #     shuffle=True, \n# #     augment=True\n# # )\n\n# # Générateur de validation (sur l'ensemble du jeu de validation NETTOYÉ)\n# # validation_generator = ScientificImageGenerator(\n# #     valid_df_clean, # <-- CORRECTION : Utiliser le DataFrame propre (322 échantillons)\n# #     batch_size=BATCH_SIZE, \n# #     img_size=IMG_SIZE, \n# #     shuffle=False, \n# #     augment=False\n# # )\n\n# # --- 2. DÉFINITION DES RAPPELS (CALLBACKS) ---\n\n# # checkpoint = ModelCheckpoint(\n# #     MODEL_NAME_PHASE1, \n# #     monitor='val_dice_coef', \n# #     verbose=1, \n# #     save_best_only=True, \n# #     mode='max'\n# # )\n# # early_stop = EarlyStopping(\n# #     monitor='val_dice_coef', \n# #     patience=1, \n# #     verbose=1, \n# #     mode='max', \n# #     restore_best_weights=True\n# # )\n# # reduce_lr = ReduceLROnPlateau(\n# #     monitor='val_dice_coef', \n# #     factor=0.5, \n# #     patience=3, \n# #     min_lr=1e-7, \n# #     verbose=1, \n# #     mode='max'\n# # )\n# # callbacks_phase1 = [checkpoint, early_stop, reduce_lr]\n\n# # --- 3. ENTRAÎNEMENT DU MODÈLE (PHASE 1) ---\n\n# # print(f\"\\n--- DÉBUT DE L'ENTRAÎNEMENT PHASE 1 (FORGÉES) ---\")\n# # print(f\"Entraînement sur {len(train_df_clean)} échantillons.\")\n\n# # history_phase1 = model.fit(\n# #     train_generator_phase1,\n# #     steps_per_epoch=len(train_df_clean) // BATCH_SIZE,\n# #     epochs=EPOCHS_PHASE1,\n# #     validation_data=validation_generator,\n# #     validation_steps=len(valid_df_clean) // BATCH_SIZE, # <-- CORRECTION : Utiliser la nouvelle taille (322)\n# #     callbacks=callbacks_phase1\n# # )\n# # print(\"--- FIN DE L'ENTRAÎNEMENT PHASE 1 ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T22:49:48.477998Z","iopub.execute_input":"2025-11-14T22:49:48.479327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import os\n# # import pandas as pd\n\n# # # --- CORRECTION DU CHEMIN D'ACCÈS ---\n# # AUTHENTIC_IMAGE_DIR = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/authentic'\n# # # -----------------------------------\n\n# # # Liste des chemins d'accès aux images authentiques\n# # authentic_paths = [\n# #     os.path.join(AUTHENTIC_IMAGE_DIR, f) \n# #     for f in os.listdir(AUTHENTIC_IMAGE_DIR) \n# #     if f.endswith(('.png', '.jpg', '.jpeg'))\n# # ]\n\n# # # Création du DataFrame 'authentic_df'\n# # authentic_data = []\n\n# # # Pour les images authentiques, le masque est vide (pas de falsification)\n# # for path in authentic_paths:\n# #     # On crée une entrée pour le masque qui sera gérée par le ScientificImageGenerator\n# #     # Si le masque est non pertinent (car c'est une image authentique), une liste vide est souvent le signal.\n# #     authentic_data.append({\n# #         'path': path,\n# #         'mask_paths': [], # Liste vide pour indiquer l'absence de masque de falsification\n# #         'case_id': os.path.basename(path).split('.')[0]\n# #     })\n\n# # # Création de la variable manquante\n# # # authentic_df = pd.DataFrame(authentic_data)\n\n# # # print(f\"✅ 'authentic_df' créé avec {len(authentic_df)} échantillons.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import pandas as pd\n# # import tensorflow as tf\n# # from tensorflow.keras.models import load_model\n\n# # # --- 1. CRÉATION DU DATAFRAME DE PHASE 2 ---\n\n# # # Concaténer les images forgées nettoyées et les images authentiques\n# # # 🚨 ATTENTION : Remplacez 'authentic_df' par le nom exact de votre DataFrame d'images authentiques !\n# # try:\n# #     train_df_phase2 = pd.concat([train_df_clean, authentic_df], ignore_index=True)\n# # except NameError:\n# #     print(\"⚠️ ERREUR : 'authentic_df' n'est pas défini. Veuillez le définir ou corriger le nom.\")\n# #     # On arrête ici si la concaténation ne peut pas se faire\n# #     # raise \n\n# # # print(f\"Taille du DataFrame Phase 2 (Forgées + Authentiques) : {len(train_df_phase2)} échantillons.\")\n# # # print(f\"Phase 2 entraînera sur {len(train_df_clean)} forgées + {len(train_df_phase2) - len(train_df_clean)} authentiques.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import tensorflow as tf\n# # from tensorflow.keras.models import load_model\n# # from tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\n# # # --- ASSUREZ-VOUS QUE VOS FONCTIONS SONT DÉFINIES EN AMONT ---\n# # # Les fonctions dice_coef et combined_loss DOIVENT être définies dans une cellule précédente\n# # # avant d'exécuter ce bloc.\n\n# # # --- PARAMÈTRES ET CHEMINS ---\n# # MODEL_NAME_PHASE1 = 'unet_resnet34_phase1.h5' # Modèle sauvegardé de la Phase 1\n# # LR_PHASE2 = 1e-6 # Taux d'apprentissage très faible (Fine-Tuning)\n# # EPOCHS_PHASE2 = 10 \n# # MODEL_NAME_PHASE2 = 'unet_resnet34_phase2_final.h5'\n# # # -----------------------------\n\n# # # 1. Chargement du modèle avec les Custom Objects Corrigés\n# # # Nous utilisons 'combined_loss' et 'dice_coef' comme défini dans votre notebook.\n# # # model_phase2 = load_model(\n# # #     MODEL_NAME_PHASE1, \n# # #     custom_objects={'combined_loss': combined_loss, 'dice_coef': dice_coef} \n# # # )\n\n# # # print(f\"✅ Modèle {MODEL_NAME_PHASE1} chargé (meilleur score Phase 1 : 0.33984).\")\n\n# # # 2. Recompilation avec un faible Learning Rate pour le Fine-Tuning\n# # # model_phase2.compile(\n# # #     optimizer=tf.keras.optimizers.Adam(learning_rate=LR_PHASE2),\n# # #     loss=combined_loss, # Utilisation de la Loss combinée\n# # #     metrics=[dice_coef, 'accuracy']\n# # # )\n\n# # # print(f\"✅ Modèle recompilé avec un Learning Rate de {LR_PHASE2} pour le Fine-Tuning.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- GÉNÉRATEUR ET CALLBACKS DE PHASE 2 ---\n\n# # Générateur d'entraînement Phase 2 (Sur toutes les données : Forgées + Authentiques)\n# # train_generator_phase2 = ScientificImageGenerator(\n# #     train_df_phase2,\n# #     batch_size=BATCH_SIZE, \n# #     img_size=IMG_SIZE, \n# #     shuffle=True, \n# #     augment=True # Conserver la Data Augmentation pour la robustesse\n# # )\n\n# # Callbacks de Fine-Tuning\n# # checkpoint_phase2 = ModelCheckpoint(\n# #     MODEL_NAME_PHASE2, \n# #     monitor='val_dice_coef', \n# #     verbose=1, \n# #     save_best_only=True, \n# #     mode='max'\n# # )\n\n# # Early Stopping très strict (patience=1) pour capturer le pic de généralisation\n# # early_stop_phase2 = EarlyStopping(\n# #     monitor='val_dice_coef', \n# #     patience=1, \n# #     verbose=1, \n# #     mode='max', \n# #     restore_best_weights=True\n# # )\n\n# # reduce_lr_phase2 = ReduceLROnPlateau(\n# #     monitor='val_dice_coef', \n# #     factor=0.5, \n# #     patience=2, # Réduction plus rapide que la Phase 1\n# #     min_lr=1e-8, \n# #     verbose=1, \n# #     mode='max'\n# # )\n# # callbacks_phase2 = [checkpoint_phase2, early_stop_phase2, reduce_lr_phase2]\n\n\n# # --- DÉMARRAGE DE L'ENTRAÎNEMENT PHASE 2 ---\n\n# # print(f\"\\n--- DÉBUT DE L'ENTRAÎNEMENT PHASE 2 (GÉNÉRALISATION vers 0.5) ---\")\n# # print(f\"Fine-Tuning sur {len(train_df_phase2)} échantillons au LR de {LR_PHASE2}.\")\n\n# # history_phase2 = model_phase2.fit(\n# #     train_generator_phase2,\n# #     steps_per_epoch=len(train_df_phase2) // BATCH_SIZE,\n# #     epochs=EPOCHS_PHASE2,\n# #     validation_data=validation_generator, # On garde le même valid_df_clean\n# #     validation_steps=len(valid_df_clean) // BATCH_SIZE,\n# #     callbacks=callbacks_phase2\n# # )\n# # print(\"--- FIN DE L'ENTRAÎNEMENT PHASE 2 ---\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # from tensorflow.keras.models import load_model\n\n# # MODEL_NAME_PHASE2 = 'unet_resnet34_phase2_final.h5'\n\n# # # 🚨 CORRECTION : Utiliser 'combined_loss' à la place de 'dice_coef_loss'\n# # # final_model = load_model(\n# # #     MODEL_NAME_PHASE2, \n# # #     custom_objects={'combined_loss': combined_loss, 'dice_coef': dice_coef} \n# # # )\n\n# # # print(\"✅ Modèle final chargé, prêt pour la prédiction.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import os\n# # import pandas as pd\n\n# # # --- Définition des chemins des données supplémentaires (confirmés) ---\n# # SUPPLEMENTAL_DIR = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/supplemental_images'\n# # SUPPLEMENTAL_MASK_DIR = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/supplemental_masks'\n\n# # # 1. Création du DataFrame supplemental_df\n# # supplemental_data = []\n# # supplemental_image_files = os.listdir(SUPPLEMENTAL_DIR)\n\n# # for img_file in supplemental_image_files:\n# #     # 1.1 Déterminer le nom de base (sans extension)\n# #     case_id, ext = os.path.splitext(img_file)\n    \n# #     # 1.2 Construire les chemins\n# #     image_path = os.path.join(SUPPLEMENTAL_DIR, img_file)\n# #     # Dans les compétitions, les masques ont souvent l'extension .png\n# #     mask_path_png = os.path.join(SUPPLEMENTAL_MASK_DIR, case_id + '.png')\n# #     mask_path_jpg = os.path.join(SUPPLEMENTAL_MASK_DIR, case_id + '.jpg')\n    \n# #     # 1.3 Vérification de l'existence du masque (plus robuste)\n# #     mask_paths = []\n# #     if os.path.exists(mask_path_png):\n# #         mask_paths.append(mask_path_png)\n# #     elif os.path.exists(mask_path_jpg):\n# #         mask_paths.append(mask_path_jpg)\n    \n# #     # 1.4 Ajout à la liste si un masque est trouvé\n# #     if mask_paths:\n# #         supplemental_data.append({\n# #             'path': image_path,\n# #             'mask_paths': mask_paths,\n# #             'case_id': case_id\n# #         })\n# #     else:\n# #         pass\n\n# # # supplemental_df = pd.DataFrame(supplemental_data)\n# # # print(f\"✅ 'supplemental_df' créé avec {len(supplemental_df)} échantillons supplémentaires (Forgés masqués).\")\n\n\n# # # 2. Concaténation avec l'ensemble Phase 2 existant\n# # # 🚨 ASSUMÉ : La variable 'train_df_phase2' (Forgées + Authentiques) est toujours définie.\n\n# # # try:\n# # #     # train_df_phase2_plus est l'ensemble total : Phase 2 + Supplémentaire\n# # #     train_df_phase2_plus = pd.concat([train_df_phase2, supplemental_df], ignore_index=True)\n# # # except NameError:\n# # #     print(\"⚠️ ERREUR : 'train_df_phase2' n'est pas défini. Veuillez relancer la cellule de concaténation Phase 2.\")\n# # #     raise\n\n# # # print(f\"Taille du DataFrame Phase 2.1 (Total) : {len(train_df_phase2_plus)} échantillons.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # import tensorflow as tf\n# # from tensorflow.keras.callbacks import ModelCheckpoint\n\n# # # --- Configuration pour l'Ultra Fine-Tuning (Phase 2.1) ---\n# # # model_phase2_1 = final_model \n# # LR_PHASE2_1 = 5e-7 # Taux d'apprentissage ultra faible (plus lent que 1e-6)\n# # EPOCHS_PHASE2_1 = 2 # Seulement 1 ou 2 époques de réajustement\n# # MODEL_NAME_PHASE2_1 = 'unet_resnet34_phase2_final_plus.h5'\n\n\n# # # 1. Recompilation du modèle (même Loss/Metrics)\n# # # model_phase2_1.compile(\n# # #     optimizer=tf.keras.optimizers.Adam(learning_rate=LR_PHASE2_1),\n# # #     loss=combined_loss,\n# # #     metrics=[dice_coef, 'accuracy']\n# # # )\n\n# # # print(f\"✅ Modèle recompilé avec un Learning Rate de {LR_PHASE2_1} pour la Phase 2.1.\")\n\n\n# # # 2. Création du générateur pour le nouvel ensemble\n# # # BATCH_SIZE, IMG_SIZE et ScientificImageGenerator sont supposés définis.\n# # # train_generator_phase2_1 = ScientificImageGenerator(\n# # #     train_df_phase2_plus,\n# # #     batch_size=BATCH_SIZE, \n# # #     img_size=IMG_SIZE, \n# # #     shuffle=True, \n# # #     augment=True \n# # # )\n\n# # # 3. Callbacks (On réutilise ceux de la Phase 2)\n# # # On utilise le nouveau nom de fichier pour la sauvegarde du meilleur modèle\n# # # checkpoint_phase2_1 = ModelCheckpoint(\n# # #     MODEL_NAME_PHASE2_1, \n# # #     monitor='val_dice_coef', \n# # #     verbose=1, \n# # #     save_best_only=True, \n# # #     mode='max'\n# # # )\n# # # On réutilise les autres callbacks de Phase 2 (early_stop_phase2, reduce_lr_phase2)\n# # # callbacks_phase2_1 = [checkpoint_phase2_1, early_stop_phase2, reduce_lr_phase2]\n\n# # # print(f\"✅ Générateur et Callbacks configurés.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- DÉMARRAGE DE L'ULTRA FINE-TUNING PHASE 2.1 ---\n\n# # print(f\"\\n--- DÉBUT DE L'ULTRA FINE-TUNING PHASE 2.1 ---\")\n# # print(f\"Fine-Tuning sur {len(train_df_phase2_plus)} échantillons au LR de {LR_PHASE2_1}.\")\n\n# # validation_generator et valid_df_clean sont supposés définis.\n\n# # history_phase2_1 = model_phase2_1.fit(\n# #     train_generator_phase2_1,\n# #     steps_per_epoch=len(train_df_phase2_plus) // BATCH_SIZE,\n# #     epochs=EPOCHS_PHASE2_1,\n# #     validation_data=validation_generator, \n# #     validation_steps=len(valid_df_clean) // BATCH_SIZE,\n# #     callbacks=callbacks_phase2_1\n# # )\n# # print(\"--- FIN DE L'ULTRA FINE-TUNING PHASE 2.1 ---\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nprint(sys.version)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T22:36:36.812948Z","iopub.execute_input":"2025-11-15T22:36:36.813227Z","iopub.status.idle":"2025-11-15T22:36:36.821936Z","shell.execute_reply.started":"2025-11-15T22:36:36.813207Z","shell.execute_reply":"2025-11-15T22:36:36.821046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FIXE CRITIQUE : Installation de la version 3.20.1 compatible Python 3.\n# Remplacez 'chemin-vers-votre-dataset' par le chemin exact.\n\n!pip install --upgrade --force-reinstall /kaggle/input/protobuf-3-20-1/protobuf-3.20.1-py2.py3-none-any.whl\n\n# (Laissez le code de désactivation du GPU)\nimport os\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nimport cv2\nimport math # Ajout de l'import math\n\nclass TestImageGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, img_size, batch_size, shuffle=False):\n        self.df = df\n        self.img_size = img_size\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        # 🚨 CORRECTION 1: Utiliser math.ceil pour inclure toutes les images \n        # (même si le dernier lot est incomplet), et garantir au moins 1 si le df n'est pas vide.\n        return int(math.ceil(len(self.df) / self.batch_size))\n\n    def __getitem__(self, index):\n        \n        indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        batch_paths = self.df.iloc[indexes]['path'].values\n        \n        # 🚨 CORRECTION 2: Initialiser X à la taille exacte du lot (len(indexes)), \n        # en particulier pour le dernier lot incomplet.\n        current_batch_size = len(indexes)\n        X = np.empty((current_batch_size, self.img_size, self.img_size, 3), dtype=np.float32)\n        \n        # Charger et prétraiter les images\n        for i, path in enumerate(batch_paths):\n            try:\n                # 1. Charger l'image\n                img = cv2.imread(path)\n                img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n                \n                # 2. Redimensionner\n                img = cv2.resize(img, (self.img_size, self.img_size))\n                \n                # 3. Normalisation (comme en entraînement)\n                img = img.astype(np.float32) / 255.0\n                \n                X[i,] = img\n                \n            except Exception as e:\n                # Gestion d'erreur CV2 existante\n                print(f\"FATAL CV2 ERROR: Chargement/Redimensionnement échoué pour l'image {path}. Remplacement par des zéros.\")\n                X[i,] = np.zeros((self.img_size, self.img_size, 3), dtype=np.float32)\n\n        return X\n\n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.df))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n\nprint(\"✅ Classe 'TestImageGenerator' corrigée et définie.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# 🚨 Vérifiez le chemin d'accès à vos images de test. Il doit être correct dans Kaggle.\nTEST_IMAGE_DIR = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images'\n\ntest_data = []\n\n# Lister tous les fichiers images de test\nfor img_file in os.listdir(TEST_IMAGE_DIR):\n    # On vérifie l'extension pour s'assurer que c'est une image\n    if img_file.endswith(('.jpg', '.jpeg', '.png')):\n        # Le 'case_id' est essentiel pour le fichier de soumission\n        case_id = img_file.split('.')[0] \n        \n        test_data.append({\n            'path': os.path.join(TEST_IMAGE_DIR, img_file),\n            'case_id': case_id \n        })\n\ntest_df = pd.DataFrame(test_data)\n\nprint(f\"✅ 'test_df' créé avec {len(test_df)} échantillons de test.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport os\nimport shutil\n\n# --- 1. Définition des chemins ---\n# Chemins d'entrée (Input Datasets)\nSEG_INPUT_PATH = '/kaggle/input/segmentation-models'\nCLS_INPUT_PATH = '/kaggle/input/classification-models'\n\n# Chemins de travail corrigés (avec underscore et dans /kaggle/working/)\nSEG_WORKING_PATH = '/kaggle/working/segmentation_models'\nCLS_WORKING_PATH = '/kaggle/working/classification_models'\n\n# --- 2. Copie et Renommage Forcé (Contournement de la Lecture Seule) ---\n\nprint(\"Début de la copie des packages vers /kaggle/working/...\")\n\n# A. Copie du package de segmentation\nif os.path.isdir(SEG_INPUT_PATH):\n    # Supprimer l'ancienne version si elle existe (pour une copie propre)\n    if os.path.exists(SEG_WORKING_PATH):\n        shutil.rmtree(SEG_WORKING_PATH)\n    \n    # Créer le nouveau dossier correctement nommé\n    os.makedirs(SEG_WORKING_PATH, exist_ok=True)\n    \n    # Copier le contenu\n    !cp -r $SEG_INPUT_PATH/* $SEG_WORKING_PATH\n    print(f\"✅ Segmentation package copié dans {SEG_WORKING_PATH}\")\nelse:\n    print(f\"⚠️ Avertissement : {SEG_INPUT_PATH} n'a pas été trouvé.\")\n\n\n# B. Copie du package de classification\nif os.path.isdir(CLS_INPUT_PATH):\n    # Supprimer l'ancienne version si elle existe\n    if os.path.exists(CLS_WORKING_PATH):\n        shutil.rmtree(CLS_WORKING_PATH)\n        \n    # Créer le nouveau dossier correctement nommé\n    os.makedirs(CLS_WORKING_PATH, exist_ok=True)\n    \n    # Copier le contenu\n    !cp -r $CLS_INPUT_PATH/* $CLS_WORKING_PATH\n    print(f\"✅ Classification package copié dans {CLS_WORKING_PATH}\")\nelse:\n    print(f\"⚠️ Avertissement : {CLS_INPUT_PATH} n'a pas été trouvé.\")\n\n\n# --- 3. Ajout des chemins corrigés à sys.path ---\n\n# Ajouter les chemins corrigés à sys.path\nif SEG_WORKING_PATH not in sys.path:\n    sys.path.append(SEG_WORKING_PATH)\n\nif CLS_WORKING_PATH not in sys.path:\n    sys.path.append(CLS_WORKING_PATH)\n\nprint(\"✅ Chemins du répertoire de travail ajoutés à sys.path.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nimport os\nimport tensorflow as tf\nimport numpy as np\n\n# --- 1. DÉFINITION DES FONCTIONS DE PERTE ET MÉTRIQUE (CUSTOM OBJECTS) ---\ndef dice_coef(y_true, y_pred, smooth=1e-7):\n    \"\"\"Coefficient de Dice (F1-score) pour la segmentation.\"\"\"\n    y_true_f = tf.keras.backend.flatten(y_true)\n    y_pred_f = tf.keras.backend.flatten(y_pred)\n    intersection = tf.keras.backend.sum(y_true_f * y_pred_f)\n    return (2. * intersection + smooth) / (tf.keras.backend.sum(y_true_f) + tf.keras.backend.sum(y_pred_f) + smooth)\n\ndef dice_loss(y_true, y_pred):\n    \"\"\"Perte de Dice.\"\"\"\n    return 1.0 - dice_coef(y_true, y_pred)\n\ndef combined_loss(y_true, y_pred, alpha=0.5):\n    \"\"\"Perte combinée (BCE + Dice Loss).\"\"\"\n    bce = tf.keras.losses.BinaryCrossentropy()\n    return alpha * dice_loss(y_true, y_pred) + (1 - alpha) * bce(y_true, y_pred)\n\nprint(\"✅ Custom objects (dice_coef, combined_loss) définis.\")\n\n\n# --- 2. GESTION DES CHEMINS (ASSUMANT LA COPIE PRÉALABLE) ---\n# Ces chemins pointent vers les dossiers dans /kaggle/working/ après la commande !cp -r\n\nCLASSIFICATION_PATH = '/kaggle/working/classification_models'  \nSEGMENTATION_PATH = '/kaggle/working/segmentation_models'  \n\nprint(\"Ajout des chemins des dépendances locales...\")\n\n# Ajouter les chemins corrigés à sys.path (uniquement s'ils ne sont pas déjà là)\nif CLASSIFICATION_PATH not in sys.path:\n    sys.path.append(CLASSIFICATION_PATH)\nif SEGMENTATION_PATH not in sys.path:\n    sys.path.append(SEGMENTATION_PATH)\n\n\n# --- 3. IMPORTATION DES PACKAGES ---\ntry:\n    # L'importation cherche dans les chemins ajoutés ci-dessus\n    import segmentation_models as sm\n    import classification_models as cm\n    print(\"✅ Packages 'segmentation_models' et 'classification_models' importés avec succès.\")\nexcept ImportError as e:\n    print(f\"❌ ÉCHEC DE L'IMPORTATION. CAUSE: {e}\")\n    \n# --- 4. DÉFINITION DES PARAMÈTRES CRUCIAUX ---\nIMG_SIZE = 256\nBATCH_SIZE = 8\nTHRESHOLD = 0.5\n\nprint(f\"\\n✅ Configuration initiale terminée. IMG_SIZE={IMG_SIZE}, BATCH_SIZE={BATCH_SIZE}, THRESHOLD={THRESHOLD}.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nfrom tensorflow.keras.optimizers import Adam # Assurez-vous d'importer l'optimiseur\n\n# Le chemin que vous avez fourni est utilisé ici pour charger le fichier .h5\nMODEL_PATH = '/kaggle/input/unet-renet34-phase2-final-plus-h5/unet_resnet34_phase2_final_plus.h5'\n\n# Définition des objets personnalisés pour le chargement\ncustom_objects = {'combined_loss': combined_loss, 'dice_coef': dice_coef}\n\n# Chargement du modèle avec les custom_objects\nfinal_model = load_model(\n    MODEL_PATH, \n    custom_objects=custom_objects\n)\n\nprint(\"✅ Modèle final chargé. Vérification de la compilation...\")\n\n# --- RECOMPILATION EXPLICITE POUR GARANTIR L'ENREGISTREMENT DES MÉTRIQUES ---\n# Ceci écrase l'avertissement et garantit que les métriques sont prêtes.\n\nfinal_model.compile(\n    # Utilisez le même optimisateur que lors de l'entraînement (par défaut Adam)\n    optimizer=Adam(learning_rate=1e-4), \n    loss=combined_loss, \n    metrics=[dice_coef] # Inclus explicitement pour l'évaluation\n)\n\nprint(\"✅ Modèle recompilé avec les pertes et métriques personnalisées.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🚨 Assurez-vous d'avoir un DataFrame 'test_df' contenant les chemins d'accès des images de test\n# ET que la classe TestImageGenerator est définie.\ntest_generator = TestImageGenerator(\n    test_df, \n    batch_size=BATCH_SIZE, \n    img_size=IMG_SIZE, \n    shuffle=False\n)\n\n# Prédiction\n# On utilise len(test_df) // BATCH_SIZE comme nombre d'étapes (steps)\nprint(f\"Début de la prédiction pour {len(test_df)} échantillons de test...\")\npredictions = final_model.predict(\n    test_generator, \n    steps=len(test_df) // BATCH_SIZE,\n    verbose=1\n)\nprint(\"✅ Prédictions terminées.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport json # Import essentiel pour le format JSON de soumission\n\ndef rle_encode(img):\n    \"\"\"\n    Encode un masque (image 2D) en Run-Length Encoding.\n    Retourne la chaîne RLE brute séparée par des espaces (ex: \"123 4 500 6\").\n    \"\"\"\n    # Doit être binaire (0 ou 1) et dans l'ordre Fortran\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    \n    # Trouve les indices de début et de fin de chaque séquence\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    \n    # Calcule la longueur de chaque séquence\n    run_lengths = runs[1::2] - runs[0::2]\n    starts = runs[0::2]\n    \n    # Zippe les départs et les longueurs en une liste plate\n    rle = list(zip(starts, run_lengths))\n    \n    # Formate en une chaîne séparée par des espaces\n    rle_string = ' '.join(str(x) for p in rle for x in p)\n    \n    return rle_string","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nimport os \n# Assurez-vous que les fonctions rle_encode, predictions, et test_df sont disponibles\n\n# --- 1. CONFIGURATION ET CHEMIN FIXÉ ---\n\n# 🚨 CHEMIN FIXÉ : Utilisation du chemin correct que vous avez fourni.\nSAMPLE_SUB_PATH = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv'\n\n# Si 'test_df' n'est pas déjà chargé (ce qui est nécessaire pour case_ids), vous devriez décommenter :\n# test_df = pd.read_csv(SAMPLE_SUB_PATH) \n\ncase_ids = test_df['case_id'].astype(str).tolist()\n\nrle_results = []\nprint(f\"Début de la conversion RLE pour {len(predictions)} masques...\")\n\n# Définition des seuils\nTHR_FIXE = 0.5             # Seuil de probabilité pour la binarisation\nMIN_PIXEL_COUNT = 100      # FILTRAGE : Masques de moins de 100 pixels ignorés\n\n\n# --- 2. TRAITEMENT, FILTRAGE, ET ENCODAGE RLE ---\nfor i, mask in enumerate(predictions):\n    # Note: Assurez-vous que 'mask' est de la taille (H, W) de l'image originale\n    case_id = case_ids[i]\n    \n    mask_binary = (mask > THR_FIXE).astype(np.uint8)\n    \n    if mask_binary.sum() < MIN_PIXEL_COUNT:\n        final_annotation = \"authentic\"\n    else:\n        # La fonction rle_encode DOIT être définie ailleurs\n        rle_encoded_str = rle_encode(mask_binary) \n        \n        if not rle_encoded_str:\n            final_annotation = \"authentic\"\n        else:\n            # Conversion en liste d'entiers puis sérialisation JSON\n            rle_list = [int(x) for x in rle_encoded_str.split(' ')]\n            final_annotation = json.dumps(rle_list)\n            \n    rle_results.append([case_id, final_annotation])\n\nprint(\"✅ Conversion RLE terminée. Format : JSON sérialisé ou authentic.\")\n\n\n# --- 3. CRÉATION DU FICHIER FINAL ---\n\nsubmission_df = pd.DataFrame(rle_results, columns=['case_id', 'annotation'])\n\ntry:\n    # Lecture du sample_submission pour garantir l'alignement de TOUS les IDs\n    ss = pd.read_csv(SAMPLE_SUB_PATH)\n    ss[\"case_id\"] = ss[\"case_id\"].astype(str)\n    submission_df[\"case_id\"] = submission_df[\"case_id\"].astype(str)\n    \n    # Fusion pour aligner et conserver tous les IDs dans le bon ordre\n    final_sub = ss[[\"case_id\"]].merge(submission_df, on=\"case_id\", how=\"left\")\n    \n    # Remplir les cas où le masque était 'None' ou non prédit par \"authentic\"\n    final_sub[\"annotation\"] = final_sub[\"annotation\"].fillna(\"authentic\")\n    \n    print(\"✅ Alignement avec sample_submission réussi.\")\n    \nexcept Exception as e:\n    # Si l'alignement échoue (ce qui ne devrait plus arriver), on utilise les résultats bruts\n    print(f\"❌ ÉCHEC FATAL lors de l'alignement : {e}\")\n    final_sub = submission_df \n\n# Sauvegarde finale\nfinal_sub.to_csv('submission.csv', index=False)\nprint(f\"✅ Fichier de soumission 'submission.csv' créé (lignes: {len(final_sub)}).\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}