{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"sourceType":"competition"},{"sourceId":14463541,"sourceType":"datasetVersion","datasetId":9235237}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook I'm going to adapt a nn-UNet to create a forgery detector in scientific images\n\n# 1. Data download, analysis, and preprocessing\n\n## 1.1 Loading libraries and data","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom glob import glob\nimport random\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport cv2\nimport shutil\nfrom PIL import Image\nfrom typing import Tuple, Union, List\n\n!pip install batchgenerators\nfrom batchgenerators.utilities.file_and_folder_operations import save_json, join\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-11T17:52:03.682363Z","iopub.execute_input":"2026-01-11T17:52:03.682542Z","iopub.status.idle":"2026-01-11T17:52:15.817702Z","shell.execute_reply.started":"2026-01-11T17:52:03.682527Z","shell.execute_reply":"2026-01-11T17:52:15.816977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Installation of our nn-UNet, the U-Net that will create 3 detectors and, when we train them, it will decide which one of those 3\n# will be our definitive detector\n!pip install nnunetv2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T12:58:59.469290Z","iopub.execute_input":"2025-11-23T12:58:59.469541Z","iopub.status.idle":"2025-11-23T12:59:03.174357Z","shell.execute_reply.started":"2025-11-23T12:58:59.469520Z","shell.execute_reply":"2025-11-23T12:59:03.173416Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Install hiddenlayer. hiddenlayer enables nnU-net to generate plots of the network topologies it generates","metadata":{}},{"cell_type":"code","source":"#pip install --upgrade git+https://github.com/FabianIsensee/hiddenlayer.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T12:59:03.175498Z","iopub.execute_input":"2025-11-23T12:59:03.175785Z","iopub.status.idle":"2025-11-23T12:59:07.855636Z","shell.execute_reply.started":"2025-11-23T12:59:03.175761Z","shell.execute_reply":"2025-11-23T12:59:07.854854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1.2 Initial exploration of the data","metadata":{}},{"cell_type":"code","source":"# Let's prepare the directories\nroot_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection'\ndirectory_train = f\"{root_dir}/train_images\"\ndirectory_masks = f\"{root_dir}/train_masks\"\ndirectory_test = f\"{root_dir}/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:11.029922Z","iopub.execute_input":"2026-01-10T19:05:11.031736Z","iopub.status.idle":"2026-01-10T19:05:11.036108Z","shell.execute_reply.started":"2026-01-10T19:05:11.031697Z","shell.execute_reply":"2026-01-10T19:05:11.035315Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's show the dataset's number of images","metadata":{}},{"cell_type":"code","source":"# Mostramos el recuento de los conjuntos de datos\nimport os\nnum_test = 0\nnum_train = 0\nnum_masks = 0\nnum_train_a = 0\nnum_train_f = 0\nfor dirname, _, filenames in os.walk(root_dir):\n    for filename in filenames:\n        if dirname.find('test_images')!= -1:\n            num_test = num_test + 1            \n        if dirname.find('train_images')!= -1:\n            num_train = num_train + 1\n            if dirname.find('authentic')!= -1:\n                num_train_a = num_train_a + 1\n            else:\n                num_train_f = num_train_f + 1\n        if dirname.find('train_masks')!= -1:\n            num_masks = num_masks + 1\nprint(\"Number of training files: {} . Forged: {}, Authentic: {}\".format(num_train,num_train_f,num_train_a))\nprint(\"Number of test files: {} \".format(num_test))\nprint(\"Number of masks: {}\".format(num_masks))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:16.974212Z","iopub.execute_input":"2026-01-10T19:05:16.974590Z","iopub.status.idle":"2026-01-10T19:05:20.595777Z","shell.execute_reply.started":"2026-01-10T19:05:16.974565Z","shell.execute_reply":"2026-01-10T19:05:20.595080Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we can see that we only have a test image, that will be used to test the model in the challenge. For this reason, to train our model in a proper way, we should divide the train dataset in train and test datasets. We have 5128 training images, 2751 of them are forged and 2377 are authentic. The quantity of masks is 2751, the same number as the forged images. This show that authentic images don't have a mask.\n\nLet's analyze the images to know how to deal with them. We will start showing some examples of images and masks.","metadata":{}},{"cell_type":"code","source":"seed = 1234\nbatch_size = 32\nclasses = os.listdir(directory_train)\ninput_size = 224","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:26.619670Z","iopub.execute_input":"2026-01-10T19:05:26.619928Z","iopub.status.idle":"2026-01-10T19:05:26.624756Z","shell.execute_reply.started":"2026-01-10T19:05:26.619911Z","shell.execute_reply":"2026-01-10T19:05:26.623852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's show the forged images","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(10,5))\npath = os.path.join(directory_train, classes[0])\nfilelist = glob(path + \"/*.png\")\nprint(f\"Class {classes[0]}\")\nfor i in range(4):\n    plt.subplot(240 + 1 + i)\n    img = plt.imread(filelist[random.randint(0,len(filelist))])\n    plt.imshow(img)\nfor i in range(4,8):\n    plt.subplot(240 + 1 + i)\n    img = plt.imread(filelist[random.randint(0,len(filelist))])\n    plt.imshow(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:31.160611Z","iopub.execute_input":"2026-01-10T19:05:31.160906Z","iopub.status.idle":"2026-01-10T19:05:33.303375Z","shell.execute_reply.started":"2026-01-10T19:05:31.160886Z","shell.execute_reply":"2026-01-10T19:05:33.302558Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now let's show the authentic images:","metadata":{}},{"cell_type":"code","source":"path = os.path.join(directory_train, classes[1])\nfilelist = glob(path + \"/*.png\")\nprint(f\"Class {classes[1]}\")\nfor i in range(4):\n    plt.subplot(240 + 1 + i)\n    img = plt.imread(filelist[random.randint(0,len(filelist))])\n    plt.imshow(img)\nfor i in range(4,8):\n    plt.subplot(240 + 1 + i)\n    img = plt.imread(filelist[random.randint(0,len(filelist))])\n    plt.imshow(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:39.255415Z","iopub.execute_input":"2026-01-10T19:05:39.255710Z","iopub.status.idle":"2026-01-10T19:05:41.071786Z","shell.execute_reply.started":"2026-01-10T19:05:39.255691Z","shell.execute_reply":"2026-01-10T19:05:41.070996Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now let's show one mask to see the format","metadata":{}},{"cell_type":"code","source":"path = os.path.normpath(directory_masks)\nfilelist = glob(path + \"/*.npy\")\nprint(f\"Masks\")\nvector = np.load(filelist[random.randint(0,len(filelist))])\nprint(vector)\nprint(vector.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:44.844937Z","iopub.execute_input":"2026-01-10T19:05:44.845247Z","iopub.status.idle":"2026-01-10T19:05:44.863517Z","shell.execute_reply.started":"2026-01-10T19:05:44.845228Z","shell.execute_reply":"2026-01-10T19:05:44.862759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now let' see if the array dimensions are the same that its corresponding image dimensions","metadata":{}},{"cell_type":"code","source":"vector = np.load(\"/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_masks/10.npy\")\nim = cv2.imread(\"/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/forged/10.png\",0)\nprint(vector.shape)\nprint(im.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:47.301420Z","iopub.execute_input":"2026-01-10T19:05:47.302090Z","iopub.status.idle":"2026-01-10T19:05:47.349381Z","shell.execute_reply.started":"2026-01-10T19:05:47.302062Z","shell.execute_reply":"2026-01-10T19:05:47.348546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"OK, now we can see that the masks have the same dimensions as the images. We can create empty masks for the authentic images in order to return the result in the training and test phases and if the array is full of 0s we will return \"Authentic\" as a result, if not we will return \"Forged\" and the mask.","metadata":{}},{"cell_type":"markdown","source":"## 1.3 Create the masks for the authentic images and put them in a new folder\n\nNow we are going to create a mask full of 0s of each authentic train images and we will storage them in an output folder called train_masks_authentic. Having masks for all the images is neccesary to train our model. To avoid working twice we will save the masks as .png, because our model needs the masks in the same file extension than the images.\n\n","metadata":{}},{"cell_type":"code","source":"# We create a function that create the folders in case they don't exist\ndef make_if_dont_exist(folder_path,overwrite=False):\n    if os.path.exists(folder_path):\n        # It exists so we tell it\n        if not overwrite:\n            print(f\"{folder_path} exists.\")\n        else:\n            print(f\"{folder_path} overwritten\")\n            shutil.rmtree(folder_path)\n            os.makedirs(folder_path)\n    else:\n      # The directory doesn't exist, so we create it\n      os.makedirs(folder_path)\n      print(f\"{folder_path} created!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:51.931933Z","iopub.execute_input":"2026-01-10T19:05:51.932229Z","iopub.status.idle":"2026-01-10T19:05:51.937427Z","shell.execute_reply.started":"2026-01-10T19:05:51.932209Z","shell.execute_reply":"2026-01-10T19:05:51.936559Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.3.1 Create the folder structure \n\nThe first step will be create our folder structure to host our training and validation images and labels. This folder structure must be adapted to the model that we will use, in our case it will be nnU-Net (we will describe it later). The specifications of nnU-Net dataset format can be found [here](https://github.com/MIC-DKFZ/nnUNet/blob/master/documentation/dataset_format.md). Each training case (image and its mask) is associated with an unique name for that case. This identifier is used by nnU-Net to connect images with the correct segmentation. In our original dataset we have authentic and forged images that share their name, for this reason we will create a new set of folders with the images and masks renamed. The image format should be {case_identifier}_0000.png for the train images and {case_identifier}.png for the labels. Our dataset has the problem that we have cases that share identifier, being some of them authentic and some forged. For this reason we have to rename our authentic masks and images making them start with an A.","metadata":{}},{"cell_type":"code","source":"# Creation of the folder were the masks will be stored\n# We will follow the folder structure that is needed for nnUNet\ndirectory_nnunet = '/kaggle/input/recod-aitrainning-dataset/nnUNet_raw' #\"/kaggle/working/nnUNet_raw\"\n#make_if_dont_exist(directory_nnunet,overwrite=False)\n\ndirectory_dataset = f\"{directory_nnunet}/Dataset555_ScientificImages\"\n#make_if_dont_exist(directory_dataset,overwrite=False)\n\ndirectory_masks = f\"{directory_dataset}/labelsTr\"\n#make_if_dont_exist(directory_masks,overwrite=False)\n\ndirectory_images = f\"{directory_dataset}/imagesTr\"\n#make_if_dont_exist(directory_images,overwrite=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T19:05:58.612800Z","iopub.execute_input":"2026-01-10T19:05:58.613320Z","iopub.status.idle":"2026-01-10T19:05:58.618989Z","shell.execute_reply.started":"2026-01-10T19:05:58.613294Z","shell.execute_reply":"2026-01-10T19:05:58.617999Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.3.2 Create the masks for the authentic images and move the images to our imagesTr folder\n\n","metadata":{}},{"cell_type":"markdown","source":"# Directory where we have the authentic images\ndirectory_train_a = f\"{directory_train}/authentic\"\n\n# Bucle to take all the authentic images from the train folder\nfor dirname, _, filenames in os.walk(directory_train_a):\n    for filename in filenames:\n        old_name = os.path.join( os.path.abspath(dirname), filename )\n\n        # Separate base from extension\n        base, extension = os.path.splitext(filename)\n\n        # get the filename without extension\n        filename_clean= os.path.basename(filename).split('.')[0]\n\n        # Initial new name\n        new_name = os.path.join(directory_images, \"A\"+filename_clean+\"_0000.png\")\n\n        # read the image without color, so it returns (x,y) vector\n        im = cv2.imread(f\"{directory_train_a}/{filename}\",0)\n        #mask = im[np.newaxis,:, :]\n        im.fill(0)\n        # save the array as an image, we will difference the authentic images making them start with an A\n        cv2.imwrite(f\"{directory_masks}/A{filename}\",im)\n\n        # Copy the original authentic image to the train test where we will have all the cases\n        # We will open it and save it to force the .png to have 3 channels\n        original_image = cv2.imread(old_name, cv2.IMREAD_COLOR)\n        original_image = cv2.cvtColor(original_image, cv2.COLOR_BGRA2BGR)\n        cv2.imwrite(new_name, original_image)\n        #mpimg.imsave(new_name, original_image)\n        #shutil.copy(old_name, new_name)\n        \n        #np.save(f\"{directory_masks_a}/{filename_clean}\",mask)","metadata":{"execution":{"iopub.status.busy":"2026-01-10T19:06:03.063154Z","iopub.execute_input":"2026-01-10T19:06:03.063414Z","iopub.status.idle":"2026-01-10T19:11:39.451891Z","shell.execute_reply.started":"2026-01-10T19:06:03.063393Z","shell.execute_reply":"2026-01-10T19:11:39.450766Z"}}},{"cell_type":"markdown","source":"Let's check","metadata":{}},{"cell_type":"markdown","source":"vector =cv2.imread(\"/kaggle/working/nnUNet_raw/Dataset555_ScientificImages/labelsTr/A1008.png\")\nim = cv2.imread(\"/kaggle/working/nnUNet_raw/Dataset555_ScientificImages/imagesTr/A1008_0000.png\")\nprint(vector.shape)\nprint(im.shape)\nplt.imshow(vector)\nplt.imshow(im)\n","metadata":{"execution":{"iopub.status.busy":"2026-01-10T19:11:39.649633Z","iopub.execute_input":"2026-01-10T19:11:39.650065Z","iopub.status.idle":"2026-01-10T19:11:39.844320Z","shell.execute_reply.started":"2026-01-10T19:11:39.650028Z","shell.execute_reply":"2026-01-10T19:11:39.843356Z"}}},{"cell_type":"markdown","source":"### 1.3.3 Move the forged images and masks to our training folders for nnUNet\n","metadata":{}},{"cell_type":"markdown","source":"#We declare the folders\ndirectory_train_forged = f\"{directory_train}/forged\"\ndirectory_masks_forged = f\"{root_dir}/train_masks\"\n\n# Bucle to copy images\nfor dirname, _, filenames in os.walk(directory_train_forged):\n    for filename in filenames:\n        old_name = os.path.join( os.path.abspath(dirname), filename )\n\n        # Separate base from extension\n        base, extension = os.path.splitext(filename)\n\n        # get the filename without extension\n        filename_clean= os.path.basename(filename).split('.')[0]\n\n        # Initial new name\n        new_name = os.path.join(directory_images, filename_clean+\"_0000.png\")\n\n        # Copy the original authentic image to the train test where we will have all the cases\n        # We will open it and save it to force the .png to have 3 channels\n        original_image = cv2.imread(old_name, cv2.IMREAD_COLOR)\n        original_image = cv2.cvtColor(original_image, cv2.COLOR_BGRA2BGR)\n        cv2.imwrite(new_name, original_image)\n\n# Bucle to move the masks\nfor dirname, _, filenames in os.walk(directory_masks_forged):\n    for filename in filenames:\n        old_name = os.path.join( os.path.abspath(dirname), filename )\n\n        # Separate base from extension\n        base, extension = os.path.splitext(filename)\n\n        # get the filename without extension\n        filename_clean= os.path.basename(filename).split('.')[0]\n\n        # Initial new name\n        new_name = os.path.join(directory_masks, filename_clean+\".png\")\n\n        vector = np.load(f\"{directory_masks_forged}/{filename}\")\n        # save the array as an image\n        cv2.imwrite(new_name,vector[0])\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2026-01-10T19:15:04.480369Z","iopub.execute_input":"2026-01-10T19:15:04.480670Z","iopub.status.idle":"2026-01-10T19:20:14.125627Z","shell.execute_reply.started":"2026-01-10T19:15:04.480624Z","shell.execute_reply":"2026-01-10T19:20:14.123873Z"}}},{"cell_type":"markdown","source":"Let's check","metadata":{}},{"cell_type":"markdown","source":"vector =cv2.imread(\"/kaggle/working/nnUNet_raw/Dataset555_ScientificImages/labelsTr/1008.png\",0)\nim = cv2.imread(\"/kaggle/working/nnUNet_raw/Dataset555_ScientificImages/imagesTr/1008_0000.png\")\nprint(vector.shape)\nprint(im.shape)\nplt.imshow(vector)\nplt.imshow(im)","metadata":{"execution":{"iopub.status.busy":"2026-01-10T19:59:09.754417Z","iopub.execute_input":"2026-01-10T19:59:09.754645Z","iopub.status.idle":"2026-01-10T19:59:09.828495Z","shell.execute_reply.started":"2026-01-10T19:59:09.754623Z","shell.execute_reply":"2026-01-10T19:59:09.827216Z"}}},{"cell_type":"markdown","source":"Now let's count how much cases we have in our dataset.","metadata":{}},{"cell_type":"code","source":"num_train = 0\nnum_masks = 0\nfor dirname, _, filenames in os.walk(directory_dataset):\n    for filename in filenames:\n        if dirname.find('imagesTr')!= -1:\n            num_train = num_train + 1  \n        if dirname.find('labelsTr')!= -1:\n            num_masks = num_masks + 1\nprint(\"Number of training files: {} . \".format(num_train))\nprint(\"Number of masks: {}\".format(num_masks))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Using nnU-Net\nNow we have the complete dataset, so we can prepare our neural network to work with it. In this notebook we will use [nnU-Net](https://github.com/MIC-DKFZ/nnUNet/tree/master). nnU-Net is a semantic segmentation method that automatically adapts to a given dataset. It will analyze the provided training cases and automatically configure a matching U-Net-based segmentation pipeline. \n\n## 2.1. Creation of the configuration file dataset.json\n\nnnU-Net needs for training a file called dataset.json, that contains the metadada of our dataset. To create it, we first have to create the function that create the dataset:","metadata":{}},{"cell_type":"markdown","source":"def generate_dataset_json(output_folder: str,\n                          channel_names: dict,\n                          labels: dict,\n                          num_training_cases: int,\n                          file_ending: str,\n                          citation: Union[List[str], str] = None,\n                          regions_class_order: Tuple[int, ...] = None,\n                          dataset_name: str = None,\n                          reference: str = None,\n                          release: str = None,\n                          description: str = None,\n                          overwrite_image_reader_writer: str = None,\n                          license: str = 'Whoever converted this dataset was lazy and didn\\'t look it up!',\n                          converted_by: str = \"Please enter your name, especially when sharing datasets with others in a common infrastructure!\",\n                          **kwargs):\n    \"\"\"\n    Generates a dataset.json file in the output folder\n\n    channel_names:\n        Channel names must map the index to the name of the channel, example:\n        {\n            0: 'T1',\n            1: 'CT'\n        }\n        Note that the channel names may influence the normalization scheme!! Learn more in the documentation.\n\n    labels:\n        This will tell nnU-Net what labels to expect. Important: This will also determine whether you use region-based training or not.\n        Example regular labels:\n        {\n            'background': 0,\n            'left atrium': 1,\n            'some other label': 2\n        }\n        Example region-based training:\n        {\n            'background': 0,\n            'whole tumor': (1, 2, 3),\n            'tumor core': (2, 3),\n            'enhancing tumor': 3\n        }\n\n        Remember that nnU-Net expects consecutive values for labels! nnU-Net also expects 0 to be background!\n\n    num_training_cases: is used to double check all cases are there!\n\n    file_ending: needed for finding the files correctly. IMPORTANT! File endings must match between images and\n    segmentations!\n\n    dataset_name, reference, release, license, description: self-explanatory and not used by nnU-Net. Just for\n    completeness and as a reminder that these would be great!\n\n    overwrite_image_reader_writer: If you need a special IO class for your dataset you can derive it from\n    BaseReaderWriter, place it into nnunet.imageio and reference it here by name\n\n    kwargs: whatever you put here will be placed in the dataset.json as well\n\n    \"\"\"\n    has_regions: bool = any([isinstance(i, (tuple, list)) and len(i) > 1 for i in labels.values()])\n    if has_regions:\n        assert regions_class_order is not None, f\"You have defined regions but regions_class_order is not set. \" \\\n                                                f\"You need that.\"\n    # channel names need strings as keys\n    keys = list(channel_names.keys())\n    for k in keys:\n        if not isinstance(k, str):\n            channel_names[str(k)] = channel_names[k]\n            del channel_names[k]\n\n    # labels need ints as values\n    for l in labels.keys():\n        value = labels[l]\n        if isinstance(value, (tuple, list)):\n            value = tuple([int(i) for i in value])\n            labels[l] = value\n        else:\n            labels[l] = int(labels[l])\n\n    dataset_json = {\n        'channel_names': channel_names,  # previously this was called 'modality'. I didn't like this so this is\n        # channel_names now. Live with it.\n        'labels': labels,\n        'numTraining': num_training_cases,\n        'file_ending': file_ending,\n        'licence': license,\n        'converted_by': converted_by\n    }\n\n    if dataset_name is not None:\n        dataset_json['name'] = dataset_name\n    if reference is not None:\n        dataset_json['reference'] = reference\n    if release is not None:\n        dataset_json['release'] = release\n    if citation is not None:\n        dataset_json['citation'] = release\n    if description is not None:\n        dataset_json['description'] = description\n    if overwrite_image_reader_writer is not None:\n        dataset_json['overwrite_image_reader_writer'] = overwrite_image_reader_writer\n    if regions_class_order is not None:\n        dataset_json['regions_class_order'] = regions_class_order\n\n    dataset_json.update(kwargs)\n\n    save_json(dataset_json, join(output_folder, 'dataset.json'), sort_keys=False)","metadata":{"execution":{"iopub.status.busy":"2025-11-23T12:59:30.456973Z","iopub.execute_input":"2025-11-23T12:59:30.457153Z","iopub.status.idle":"2025-11-23T12:59:30.467156Z","shell.execute_reply.started":"2025-11-23T12:59:30.457139Z","shell.execute_reply":"2025-11-23T12:59:30.466319Z"}}},{"cell_type":"markdown","source":"Now we can use the function and store our dataset.json in our dataset folder.","metadata":{}},{"cell_type":"markdown","source":"#We declare the variables\nnum_training_cases= 5128 \ndataset_name = 'Dataset555_ScientificImages'\n\n#We run the function\ngenerate_dataset_json(directory_dataset, {0: 'R', 1: 'G', 2: 'B'}, {'background': 0, 'forgery': 1},\n                          num_training_cases, '.png', dataset_name=dataset_name)\n","metadata":{}},{"cell_type":"markdown","source":"Now we have the .json archive created, so let's verify our dataset integrity","metadata":{}},{"cell_type":"code","source":"#Lets create nnU-Net folders\nmount_dir = '/kaggle/input/recod-aitrainning-dataset'\nworking_dir =  '/kaggle/working'\ntemp_dir ='/kaggle/tmp/'\ndirectory_preprocessed = f\"{temp_dir}/nnUNet_preprocessed\"\nmake_if_dont_exist(directory_preprocessed,overwrite=False)\ndirectory_results = f\"{working_dir}/nnUNet_Results_Folder\"\nmake_if_dont_exist(directory_results,overwrite=False)\n\n# Creation of environment variables\n\nprint(\"Current Working Directory {}\".format(os.getcwd()))\npath_dict = {\n    \"nnUNet_raw\" : os.path.join(mount_dir, \"nnUNet_raw\"),\n    \"nnUNet_preprocessed\" : os.path.join(temp_dir, \"nnUNet_preprocessed\"),\n    \"nnUNet_results\" : os.path.join(working_dir, \"nnUNet_Results_Folder\"),\n}\n# Write paths to environment variables\nfor env_var, path in path_dict.items():\n  os.environ[env_var] = path\n\n# Check whether all environment variables are set correct!\nfor env_var, path in path_dict.items():\n  if os.getenv(env_var) != path:\n    print(\"Error:\")\n    print(\"Environment Variable {} is not set correctly!\".format(env_var))\n    print(\"Should be {}\".format(path))\n    print(\"Variable is {}\".format(os.getenv(env_var)))\n  make_if_dont_exist(path, overwrite=False)\n\n!export nnUNet_raw=os.path.join(mount_dir, \"nnUNet_raw\")\n!export nnUNet_preprocessed=os.path.join(temp_dir, \"nnUNet_preprocessed\")\n!export nnUNet_results=os.path.join(working_dir, \"nnUNet_Results_Folder\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T12:59:30.467880Z","iopub.execute_input":"2025-11-23T12:59:30.468181Z","iopub.status.idle":"2025-11-23T12:59:30.834453Z","shell.execute_reply.started":"2025-11-23T12:59:30.468155Z","shell.execute_reply":"2025-11-23T12:59:30.833752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We run the dataset integrity verification\n!nnUNetv2_plan_and_preprocess -d 555 \n#--verify_dataset_integrity","metadata":{"execution":{"iopub.status.busy":"2025-11-21T19:31:50.060375Z","iopub.execute_input":"2025-11-21T19:31:50.060614Z","iopub.status.idle":"2025-11-21T19:54:23.486588Z","shell.execute_reply.started":"2025-11-21T19:31:50.060594Z","shell.execute_reply":"2025-11-21T19:54:23.485758Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.2. Trainning the model\n\nNow that we have everything, let's train out model. ","metadata":{}},{"cell_type":"code","source":"!nnUNetv2_train 555 2d 0 -tr nnUNetTrainer_100epochs --npz ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!nnUNetv2_train 555 2d 1 -tr nnUNetTrainer_100epochs --npz ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!nnUNetv2_train 555 2d 2 -tr nnUNetTrainer_250epochs --npz ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T12:59:30.835620Z","iopub.execute_input":"2025-11-23T12:59:30.835909Z","iopub.status.idle":"2025-11-23T12:59:57.742847Z","shell.execute_reply.started":"2025-11-23T12:59:30.835886Z","shell.execute_reply":"2025-11-23T12:59:57.742122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!nnUNetv2_train 555 2d 3 -tr nnUNetTrainer_250epochs --npz ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-23T12:59:57.743766Z","iopub.execute_input":"2025-11-23T12:59:57.743976Z","iopub.status.idle":"2025-11-23T13:00:06.348934Z","shell.execute_reply.started":"2025-11-23T12:59:57.743957Z","shell.execute_reply":"2025-11-23T13:00:06.347994Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# !nnUNetv2_train 555 2d 4 -tr nnUNetTrainer_250epochs --npz ","metadata":{"execution":{"iopub.status.busy":"2025-11-23T13:00:06.350089Z","iopub.execute_input":"2025-11-23T13:00:06.350443Z","iopub.status.idle":"2025-11-23T13:00:10.461116Z","shell.execute_reply.started":"2025-11-23T13:00:06.350420Z","shell.execute_reply":"2025-11-23T13:00:10.460381Z"}}},{"cell_type":"markdown","source":"## 2.3. Finding the best configuration\n\nWe have trained 3 models 5 times, let's find which configuration is the best one","metadata":{}},{"cell_type":"code","source":"!nnUNetv2_find_best_configuration 555 -f 0 -c 2d -tr nnUNetTrainer_100epochs","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.4. Predictions with our best configuration\n\nNow we are going to use our best configuration to obtain the prediction of the test image. To be used by nnUNet, the image name must end in _0000","metadata":{}},{"cell_type":"code","source":"directory_output = '/kaggle/working/final_prediction'\nmake_if_dont_exist(directory_output,overwrite=False)\n\n#We transform the test image to have 3 channels\ndirectory_img_test = '/kaggle/working/img_test'\nmake_if_dont_exist(directory_img_test,overwrite=False)\n# Copy the original authentic image to the train test where we will have all the cases\n# We will open it and save it to force the .png to have 3 channels\noriginal_image = cv2.imread(\"/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images/45.png\", cv2.IMREAD_COLOR)\noriginal_image = cv2.cvtColor(original_image, cv2.COLOR_BGRA2BGR)\nnew_name = os.path.join(directory_img_test, \"45_0000.png\")\ncv2.imwrite(new_name, original_image)\n\n!nnUNetv2_predict -d Dataset555_ScientificImages -i \"/kaggle/working/img_test\" -o \"/kaggle/working/final_prediction\" -f 0 -tr nnUNetTrainer_100epochs -c 2d -p nnUNetPlans\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we have our prediction, so we will create the submission csv. For that we will have to check if the file is an empty mask or not to infer if the image is authentic or forged\n","metadata":{}},{"cell_type":"code","source":"# We create the dataframe\ndt_submission = pd.DataFrame(np.nan,index=[], columns=['case_id','annotation'])\n\nfor images in os.listdir(directory_output):\n    # check if the image ends with png\n    if (images.endswith('.png')):\n        #We have the image, let's transform to an array\n        # read the image without color, so it returns (x,y) vector\n        im = cv2.imread(f\"{directory_output}/{images}\",0)\n        mask = im[np.newaxis,:, :]\n        if np.count_nonzero(mask) == 0:\n            # We append the result -> the image is authentic\n            dt_submission.loc[len(dt_submission)] = ['45', 'authentic']\n        else:\n            # We append the result -> the image is forged\n            dt_submission.loc[len(dt_submission)] = ['45', 'forged']\n\n#Let's export the dataframe to csv\ndt_submission.to_csv(os.path.join(directory_output,'submission.csv'), index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And let's show the mask","metadata":{}},{"cell_type":"code","source":"mask_predicted = \"/kaggle/working/final_prediction/45.png\"\nvector = cv2.imread(mask_predicted,0)\nplt.imshow(vector) ","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}