{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":14346210,"sourceType":"datasetVersion","datasetId":9160144},{"sourceId":14353530,"sourceType":"datasetVersion","datasetId":9165180}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pathlib\nimport PIL\n\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:06:43.027496Z","iopub.execute_input":"2026-01-08T13:06:43.028087Z","iopub.status.idle":"2026-01-08T13:07:12.472204Z","shell.execute_reply.started":"2026-01-08T13:06:43.027862Z","shell.execute_reply":"2026-01-08T13:07:12.471289Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Database presentation","metadata":{}},{"cell_type":"code","source":"path = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.474474Z","iopub.execute_input":"2026-01-08T13:07:12.475343Z","iopub.status.idle":"2026-01-08T13:07:12.480284Z","shell.execute_reply.started":"2026-01-08T13:07:12.475304Z","shell.execute_reply":"2026-01-08T13:07:12.479227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_dir = pathlib.Path(path)\ntrain_images_dir = base_dir/\"train_images\"\ntrain_masks_dir = base_dir/\"train_masks\"\nauthentic_dir = train_images_dir/\"authentic\"\nforged_dir = train_images_dir/\"forged\"\n\nsupplemental_img_dir = base_dir/\"supplemental_images\"\nsupplemental_masks_masks = base_dir/\"supplemental_masks\"\n\ntest_dir = base_dir/\"test_images\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.481761Z","iopub.execute_input":"2026-01-08T13:07:12.482114Z","iopub.status.idle":"2026-01-08T13:07:12.509080Z","shell.execute_reply.started":"2026-01-08T13:07:12.482083Z","shell.execute_reply":"2026-01-08T13:07:12.508051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# images_dataset = []\n\n# # 1 -> authentic image\n# # 0 -> forged image\n# # 2 -> masks image\n\n# for i in authentic_dir.iterdir():\n#     images_dataset.append({\"path\":i,\"label\":1})\n\n# for i in forged_dir.iterdir():\n#     images_dataset.append({\"path\":i,\"label\":0})\n\n# for i in train_masks_dir.iterdir():\n#     images_dataset.append({\"path\":i,\"label\":2})\n\n# df = pd.DataFrame(images_dataset)\n# df[\"label\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.510175Z","iopub.execute_input":"2026-01-08T13:07:12.510526Z","iopub.status.idle":"2026-01-08T13:07:12.532573Z","shell.execute_reply.started":"2026-01-08T13:07:12.510493Z","shell.execute_reply":"2026-01-08T13:07:12.531440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.534146Z","iopub.execute_input":"2026-01-08T13:07:12.535620Z","iopub.status.idle":"2026-01-08T13:07:12.561348Z","shell.execute_reply.started":"2026-01-08T13:07:12.535583Z","shell.execute_reply":"2026-01-08T13:07:12.560141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_sample = df[df[\"label\"] == 1].sample(n=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.562815Z","iopub.execute_input":"2026-01-08T13:07:12.563223Z","iopub.status.idle":"2026-01-08T13:07:12.586952Z","shell.execute_reply.started":"2026-01-08T13:07:12.563183Z","shell.execute_reply":"2026-01-08T13:07:12.585885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.590222Z","iopub.execute_input":"2026-01-08T13:07:12.590546Z","iopub.status.idle":"2026-01-08T13:07:12.616734Z","shell.execute_reply.started":"2026-01-08T13:07:12.590517Z","shell.execute_reply":"2026-01-08T13:07:12.615183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for sample in train_sample[\"path\"]:\n#     img = PIL.Image.open(sample)\n#     plt.figure(figsize=(10,10))\n#     plt.subplot(1, 3, 1)\n#     plt.imshow(img)\n#     plt.title(\"Authentic\")\n#     plt.axis(\"off\")\n\n#     img_name = sample.stem\n\n#     forged_path = forged_dir/str(img_name+sample.suffix)\n#     forged_img = PIL.Image.open(forged_path)\n#     plt.subplot(1, 3, 2)\n#     plt.imshow(forged_img)\n#     plt.title(\"Forged\")\n#     plt.axis(\"off\")\n\n#     plt.subplot(1, 3, 3)\n#     plt.imshow(forged_img)\n#     mask_path = train_masks_dir/ f\"{img_name}.npy\"\n\n#     if mask_path.exists():\n#         mask = np.load(mask_path)\n#         # print(mask)\n#         mask = np.array(mask, dtype=np.float32)[0]\n#         mask = (mask>0)\n#         # print(mask)\n#         plt.imshow(mask, cmap=\"Reds\", alpha=0.7)\n#         plt.title(\"Forged Area (Red)\")\n#     else:\n#         plt.title(\"Mask not found\")\n#     plt.axis(\"off\")\n#     plt.show()\n#     # print(sample)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.618590Z","iopub.execute_input":"2026-01-08T13:07:12.619031Z","iopub.status.idle":"2026-01-08T13:07:12.643490Z","shell.execute_reply.started":"2026-01-08T13:07:12.619000Z","shell.execute_reply":"2026-01-08T13:07:12.641130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df = df[df[\"label\"]  == 0].reset_index(drop=True)\n# df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.644795Z","iopub.execute_input":"2026-01-08T13:07:12.645423Z","iopub.status.idle":"2026-01-08T13:07:12.677593Z","shell.execute_reply.started":"2026-01-08T13:07:12.645388Z","shell.execute_reply":"2026-01-08T13:07:12.676636Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## supplemental dataset","metadata":{}},{"cell_type":"code","source":"# images_dataset = []\n\n# # 1 -> authentic image\n# # 0 -> forged image\n# # 2 -> masks image\n\n# for i in supplemental_img_dir.iterdir():\n#     images_dataset.append({\"path\":i,\"label\":0})\n\n# for i in supplemental_masks_masks.iterdir():\n#     images_dataset.append({\"path\":i,\"label\":2})\n\n# new_df = pd.DataFrame(images_dataset)\n# new_df[\"label\"].value_counts()\n# new_df = new_df[new_df[\"label\"]  == 0].reset_index(drop=True)\n# new_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.678698Z","iopub.execute_input":"2026-01-08T13:07:12.679054Z","iopub.status.idle":"2026-01-08T13:07:12.703390Z","shell.execute_reply.started":"2026-01-08T13:07:12.679026Z","shell.execute_reply":"2026-01-08T13:07:12.701759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"# from PIL import Image\n# import numpy as np\n# from pathlib import Path\n\n# def PreprocessData(df):\n#     x_pro = []\n#     y_pro = []\n\n#     for file in df[\"path\"]:\n#         file = Path(file)\n\n#         # ---- Load image ----\n#         img = Image.open(file).convert(\"RGB\")\n#         img = img.resize((256, 256))\n#         img = np.array(img, dtype=np.float32) / 255.0\n#         x_pro.append(img)\n\n#         # ---- Load mask ----\n#         mask_path = train_masks_dir / f\"{file.stem}.npy\"\n#         mask = np.load(mask_path)\n\n#         # Handle mask shape safely\n#         if mask.ndim == 3:\n#             mask = mask[0]\n\n#         mask = Image.fromarray(mask)\n#         mask = mask.resize((256, 256), resample=Image.NEAREST)\n#         mask = np.array(mask, dtype=np.float32)\n#         mask = np.expand_dims(mask, axis=-1)\n\n#         y_pro.append(mask)\n\n#     return np.array(x_pro), np.array(y_pro)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.704514Z","iopub.execute_input":"2026-01-08T13:07:12.704792Z","iopub.status.idle":"2026-01-08T13:07:12.731182Z","shell.execute_reply.started":"2026-01-08T13:07:12.704765Z","shell.execute_reply":"2026-01-08T13:07:12.729182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# x,y = PreprocessData(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.732549Z","iopub.execute_input":"2026-01-08T13:07:12.732916Z","iopub.status.idle":"2026-01-08T13:07:12.784144Z","shell.execute_reply.started":"2026-01-08T13:07:12.732875Z","shell.execute_reply":"2026-01-08T13:07:12.782912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from PIL import Image\n# import numpy as np\n# from pathlib import Path\n\n# def PreprocessData_2(df):\n#     x_pro = []\n#     y_pro = []\n\n#     for file in df[\"path\"]:\n#         file = Path(file)\n\n#         # ---- Load image ----\n#         img = Image.open(file).convert(\"RGB\")\n#         img = img.resize((256, 256))\n#         img = np.array(img, dtype=np.float32) / 255.0\n#         x_pro.append(img)\n\n#         # ---- Load mask ----\n#         mask_path = supplemental_masks_masks / f\"{file.stem}.npy\"\n#         mask = np.load(mask_path)\n\n#         # Handle mask shape safely\n#         if mask.ndim == 3:\n#             mask = mask[0]\n\n#         mask = Image.fromarray(mask)\n#         mask = mask.resize((256, 256), resample=Image.NEAREST)\n#         mask = np.array(mask, dtype=np.float32)\n#         mask = np.expand_dims(mask, axis=-1)\n\n#         y_pro.append(mask)\n\n#     return np.array(x_pro), np.array(y_pro)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.785433Z","iopub.execute_input":"2026-01-08T13:07:12.785767Z","iopub.status.idle":"2026-01-08T13:07:12.811537Z","shell.execute_reply.started":"2026-01-08T13:07:12.785728Z","shell.execute_reply":"2026-01-08T13:07:12.810492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# x2,y2 = PreprocessData_2(new_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.812613Z","iopub.execute_input":"2026-01-08T13:07:12.812951Z","iopub.status.idle":"2026-01-08T13:07:12.836058Z","shell.execute_reply.started":"2026-01-08T13:07:12.812897Z","shell.execute_reply":"2026-01-08T13:07:12.834870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# x = np.concatenate([x, x2], axis=0)\n# y = np.concatenate([y, y2], axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.837440Z","iopub.execute_input":"2026-01-08T13:07:12.837792Z","iopub.status.idle":"2026-01-08T13:07:12.857780Z","shell.execute_reply.started":"2026-01-08T13:07:12.837755Z","shell.execute_reply":"2026-01-08T13:07:12.856165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (x.shape,y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.859587Z","iopub.execute_input":"2026-01-08T13:07:12.860362Z","iopub.status.idle":"2026-01-08T13:07:12.882753Z","shell.execute_reply.started":"2026-01-08T13:07:12.860130Z","shell.execute_reply":"2026-01-08T13:07:12.881647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model presentation","metadata":{}},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow.keras import layers, models\n# from tensorflow.keras.applications import ResNet50","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.884402Z","iopub.execute_input":"2026-01-08T13:07:12.884682Z","iopub.status.idle":"2026-01-08T13:07:12.904399Z","shell.execute_reply.started":"2026-01-08T13:07:12.884657Z","shell.execute_reply":"2026-01-08T13:07:12.903270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def attention_gate(skip, gating, inter_channels):\n#     theta_x = layers.Conv2D(inter_channels, 1, padding=\"same\")(skip)\n#     phi_g = layers.Conv2D(inter_channels, 1, padding=\"same\")(gating)\n\n#     add = layers.Add()([theta_x, phi_g])\n#     add = layers.Activation(\"relu\")(add)\n\n#     psi = layers.Conv2D(1, 1, padding=\"same\")(add)\n#     psi = layers.Activation(\"sigmoid\")(psi)\n\n#     return layers.Multiply()([skip, psi])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.905855Z","iopub.execute_input":"2026-01-08T13:07:12.906224Z","iopub.status.idle":"2026-01-08T13:07:12.932200Z","shell.execute_reply.started":"2026-01-08T13:07:12.906197Z","shell.execute_reply":"2026-01-08T13:07:12.931173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def decoder_block(x, skip, filters):\n#     x = layers.Conv2DTranspose(filters, 2, strides=2, padding=\"same\")(x)\n\n#     skip = attention_gate(skip, x, filters // 2)\n\n#     x = layers.Concatenate()([x, skip])\n\n#     x = layers.Conv2D(filters, 3, padding=\"same\", activation=\"relu\")(x)\n#     x = layers.Conv2D(filters, 3, padding=\"same\", activation=\"relu\")(x)\n\n#     return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.933416Z","iopub.execute_input":"2026-01-08T13:07:12.933713Z","iopub.status.idle":"2026-01-08T13:07:12.953226Z","shell.execute_reply.started":"2026-01-08T13:07:12.933687Z","shell.execute_reply":"2026-01-08T13:07:12.952010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def ResNet50_Attention_UNet(input_shape=(256,256,3)):\n#     inputs = layers.Input(input_shape)\n\n#     # Pretrained ResNet50 encoder\n#     base = ResNet50(\n#         include_top=False,\n#         weights=\"/kaggle/input/imagenet-resnet50-weights/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5\",\n#         input_tensor=inputs\n#     )\n\n#     # Encoder feature maps (skip connections)\n#     s1 = base.get_layer(\"conv1_relu\").output          # 128×128\n#     s2 = base.get_layer(\"conv2_block3_out\").output    # 64×64\n#     s3 = base.get_layer(\"conv3_block4_out\").output    # 32×32\n#     s4 = base.get_layer(\"conv4_block6_out\").output    # 16×16\n\n#     bottleneck = base.get_layer(\"conv5_block3_out\").output  # 8×8\n\n#     # Decoder\n#     d1 = decoder_block(bottleneck, s4, 512)   # 8 → 16\n#     d2 = decoder_block(d1, s3, 256)           # 16 → 32\n#     d3 = decoder_block(d2, s2, 128)           # 32 → 64\n#     d4 = decoder_block(d3, s1, 64)             # 64 → 128\n#     d5 = layers.Conv2DTranspose(32, 2, strides=2, padding=\"same\")(d4)\n#     d5 = layers.Conv2D(32, 3, padding=\"same\", activation=\"relu\")(d5)\n#     d5 = layers.Conv2D(32, 3, padding=\"same\", activation=\"relu\")(d5)\n\n#     outputs = layers.Conv2D(1, 1, activation=\"sigmoid\")(d5)\n#     model = models.Model(inputs, outputs, name=\"ResNet50_Attention_UNet\")\n#     return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.954726Z","iopub.execute_input":"2026-01-08T13:07:12.955141Z","iopub.status.idle":"2026-01-08T13:07:12.974463Z","shell.execute_reply.started":"2026-01-08T13:07:12.955060Z","shell.execute_reply":"2026-01-08T13:07:12.973291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def dice_loss(y_true, y_pred, smooth=1e-6):\n#     intersection = tf.reduce_sum(y_true * y_pred)\n#     return 1 - (2 * intersection + smooth) / (\n#         tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) + smooth\n#     )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:12.975550Z","iopub.execute_input":"2026-01-08T13:07:12.975913Z","iopub.status.idle":"2026-01-08T13:07:13.002683Z","shell.execute_reply.started":"2026-01-08T13:07:12.975877Z","shell.execute_reply":"2026-01-08T13:07:13.001299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def focal_loss(y_true, y_pred, gamma=2.0, alpha=0.25):\n#     y_pred = tf.clip_by_value(y_pred, 1e-7, 1 - 1e-7)\n#     pt = tf.where(tf.equal(y_true, 1), y_pred, 1 - y_pred)\n#     return -tf.reduce_mean(alpha * tf.pow(1 - pt, gamma) * tf.math.log(pt))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.008296Z","iopub.execute_input":"2026-01-08T13:07:13.008608Z","iopub.status.idle":"2026-01-08T13:07:13.028480Z","shell.execute_reply.started":"2026-01-08T13:07:13.008582Z","shell.execute_reply":"2026-01-08T13:07:13.027134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def combined_loss(y_true, y_pred):\n#     return dice_loss(y_true, y_pred) + focal_loss(y_true, y_pred)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.030369Z","iopub.execute_input":"2026-01-08T13:07:13.030713Z","iopub.status.idle":"2026-01-08T13:07:13.050720Z","shell.execute_reply.started":"2026-01-08T13:07:13.030683Z","shell.execute_reply":"2026-01-08T13:07:13.049695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# callbacks = [\n#     tf.keras.callbacks.ModelCheckpoint(\n#         filepath=\"resnet_attention_unet_best.h5\",\n#         monitor=\"val_loss\",\n#         save_best_only=True,\n#         verbose=1\n#     ),\n\n#     tf.keras.callbacks.EarlyStopping(\n#         monitor=\"val_loss\",\n#         patience=10,\n#         restore_best_weights=True,\n#         verbose=1\n#     ),\n\n#     tf.keras.callbacks.ReduceLROnPlateau(\n#         monitor=\"val_loss\",\n#         factor=0.3,\n#         patience=5,\n#         min_lr=1e-6,\n#         verbose=1\n#     )\n# ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.052446Z","iopub.execute_input":"2026-01-08T13:07:13.052785Z","iopub.status.idle":"2026-01-08T13:07:13.074690Z","shell.execute_reply.started":"2026-01-08T13:07:13.052759Z","shell.execute_reply":"2026-01-08T13:07:13.073633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = ResNet50_Attention_UNet()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.075944Z","iopub.execute_input":"2026-01-08T13:07:13.076286Z","iopub.status.idle":"2026-01-08T13:07:13.101692Z","shell.execute_reply.started":"2026-01-08T13:07:13.076257Z","shell.execute_reply":"2026-01-08T13:07:13.100744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for layer in model.layers:\n#     if \"resnet\" in layer.name.lower():\n#         layer.trainable = False\n\n# for layer in model.layers[-40:]:\n#     layer.trainable = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.103067Z","iopub.execute_input":"2026-01-08T13:07:13.103360Z","iopub.status.idle":"2026-01-08T13:07:13.127921Z","shell.execute_reply.started":"2026-01-08T13:07:13.103335Z","shell.execute_reply":"2026-01-08T13:07:13.126758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model.compile(\n#     optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n#     loss=combined_loss,\n#     metrics=[\"accuracy\"]\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.129171Z","iopub.execute_input":"2026-01-08T13:07:13.129515Z","iopub.status.idle":"2026-01-08T13:07:13.146968Z","shell.execute_reply.started":"2026-01-08T13:07:13.129480Z","shell.execute_reply":"2026-01-08T13:07:13.146032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# x_train,x_test,y_train,y_test = train_test_split(x,y,test_size=0.2,random_state=42)\n# (x_train.shape,y_train.shape),(x_test.shape,y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.148060Z","iopub.execute_input":"2026-01-08T13:07:13.148366Z","iopub.status.idle":"2026-01-08T13:07:13.170153Z","shell.execute_reply.started":"2026-01-08T13:07:13.148333Z","shell.execute_reply":"2026-01-08T13:07:13.168312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# history = model.fit(\n#     x_train, y_train,\n#     validation_data=(x_test, y_test),\n#     epochs=25,\n#     batch_size=4,\n#     callbacks=callbacks\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.171651Z","iopub.execute_input":"2026-01-08T13:07:13.172000Z","iopub.status.idle":"2026-01-08T13:07:13.195662Z","shell.execute_reply.started":"2026-01-08T13:07:13.171969Z","shell.execute_reply":"2026-01-08T13:07:13.193387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n\n# plt.figure(figsize=(12,4))\n\n# plt.subplot(1,2,1)\n# plt.plot(history.history[\"loss\"], label=\"train\")\n# plt.plot(history.history[\"val_loss\"], label=\"val\")\n# plt.title(\"Loss\")\n# plt.legend()\n\n# plt.subplot(1,2,2)\n# plt.plot(history.history[\"accuracy\"], label=\"train\")\n# plt.plot(history.history[\"val_accuracy\"], label=\"val\")\n# plt.title(\"Accuracy\")\n# plt.legend()\n\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.197317Z","iopub.execute_input":"2026-01-08T13:07:13.197793Z","iopub.status.idle":"2026-01-08T13:07:13.219403Z","shell.execute_reply.started":"2026-01-08T13:07:13.197754Z","shell.execute_reply":"2026-01-08T13:07:13.218407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def show_prediction(model, x, y, idx=0):\n#     pred = model.predict(x[idx:idx+1])[0]\n\n#     plt.figure(figsize=(12,4))\n\n#     plt.subplot(1,3,1)\n#     plt.imshow(x[idx])\n#     plt.title(\"Image\")\n#     plt.axis(\"off\")\n\n#     plt.subplot(1,3,2)\n#     plt.imshow(y[idx].squeeze(), cmap=\"gray\")\n#     plt.title(\"Ground Truth\")\n#     plt.axis(\"off\")\n\n#     plt.subplot(1,3,3)\n#     plt.imshow(pred.squeeze()>0.25, cmap=\"gray\")\n#     plt.title(\"Prediction\")\n#     plt.axis(\"off\")\n#     # plt.save()\n#     plt.show()\n#     return pred.squeeze()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.220672Z","iopub.execute_input":"2026-01-08T13:07:13.221056Z","iopub.status.idle":"2026-01-08T13:07:13.243390Z","shell.execute_reply.started":"2026-01-08T13:07:13.221024Z","shell.execute_reply":"2026-01-08T13:07:13.242288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Save Model","metadata":{}},{"cell_type":"code","source":"ResNet50_Attention_UNet_Model = \"/kaggle/input/resnet-attention-unet-best/resnet_attention_unet_best.h5\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.244291Z","iopub.execute_input":"2026-01-08T13:07:13.244569Z","iopub.status.idle":"2026-01-08T13:07:13.272363Z","shell.execute_reply.started":"2026-01-08T13:07:13.244545Z","shell.execute_reply":"2026-01-08T13:07:13.271000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def dice_loss(y_true, y_pred, smooth=1e-6):\n    intersection = tf.reduce_sum(y_true * y_pred)\n    return 1 - (2 * intersection + smooth) / (\n        tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) + smooth\n    )\ndef focal_loss(y_true, y_pred, gamma=2.0, alpha=0.25):\n    y_pred = tf.clip_by_value(y_pred, 1e-7, 1 - 1e-7)\n    pt = tf.where(tf.equal(y_true, 1), y_pred, 1 - y_pred)\n    return -tf.reduce_mean(alpha * tf.pow(1 - pt, gamma) * tf.math.log(pt))\ndef combined_loss(y_true, y_pred):\n    return dice_loss(y_true, y_pred) + focal_loss(y_true, y_pred)\n\ndef ResNet50_Attention_UNet(input_shape=(256,256,3)):\n    inputs = layers.Input(input_shape)\n\n    # Pretrained ResNet50 encoder\n    base = ResNet50(\n        include_top=False,\n        weights=\"/kaggle/input/imagenet-resnet50-weights/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5\",\n        input_tensor=inputs\n    )\n\n    # Encoder feature maps (skip connections)\n    s1 = base.get_layer(\"conv1_relu\").output          # 128×128\n    s2 = base.get_layer(\"conv2_block3_out\").output    # 64×64\n    s3 = base.get_layer(\"conv3_block4_out\").output    # 32×32\n    s4 = base.get_layer(\"conv4_block6_out\").output    # 16×16\n\n    bottleneck = base.get_layer(\"conv5_block3_out\").output  # 8×8\n\n    # Decoder\n    d1 = decoder_block(bottleneck, s4, 512)   # 8 → 16\n    d2 = decoder_block(d1, s3, 256)           # 16 → 32\n    d3 = decoder_block(d2, s2, 128)           # 32 → 64\n    d4 = decoder_block(d3, s1, 64)             # 64 → 128\n    d5 = layers.Conv2DTranspose(32, 2, strides=2, padding=\"same\")(d4)\n    d5 = layers.Conv2D(32, 3, padding=\"same\", activation=\"relu\")(d5)\n    d5 = layers.Conv2D(32, 3, padding=\"same\", activation=\"relu\")(d5)\n\n    outputs = layers.Conv2D(1, 1, activation=\"sigmoid\")(d5)\n    model = models.Model(inputs, outputs, name=\"ResNet50_Attention_UNet\")\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.273684Z","iopub.execute_input":"2026-01-08T13:07:13.274091Z","iopub.status.idle":"2026-01-08T13:07:13.297088Z","shell.execute_reply.started":"2026-01-08T13:07:13.274052Z","shell.execute_reply":"2026-01-08T13:07:13.295982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = tf.keras.models.load_model(\n    \"/kaggle/input/resnet-attention-unet-best/resnet_attention_unet_best.h5\",\n    custom_objects={\n        \"dice_loss\": dice_loss,\n        \"focal_loss\": focal_loss,\n        \"combined_loss\": combined_loss,\n        \"ResNet50_Attention_UNet\": ResNet50_Attention_UNet,\n    },\n    compile=False   \n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:13.298454Z","iopub.execute_input":"2026-01-08T13:07:13.298769Z","iopub.status.idle":"2026-01-08T13:07:17.423578Z","shell.execute_reply.started":"2026-01-08T13:07:13.298734Z","shell.execute_reply":"2026-01-08T13:07:17.422295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"# import json\n\n# import numba\n# import numpy as np\n# from numba import types\n# import numpy.typing as npt\n# import pandas as pd\n# import scipy.optimize\n\n\n# class ParticipantVisibleError(Exception):\n#     pass\n\n\n# @numba.jit(nopython=True)\n# def _rle_encode_jit(x: npt.NDArray, fg_val: int = 1) -> list[int]:\n#     \"\"\"Numba-jitted RLE encoder.\"\"\"\n#     dots = np.where(x.T.flatten() == fg_val)[0]\n#     run_lengths = []\n#     prev = -2\n#     for b in dots:\n#         if b > prev + 1:\n#             run_lengths.extend((b + 1, 0))\n#         run_lengths[-1] += 1\n#         prev = b\n#     return run_lengths\n\n\n# def rle_encode(masks: list[npt.NDArray], fg_val: int = 1) -> str:\n#     \"\"\"\n#     Adapted from contrails RLE https://www.kaggle.com/code/inversion/contrails-rle-submission\n#     Args:\n#         masks: list of numpy array of shape (height, width), 1 - mask, 0 - background\n#     Returns: run length encodings as a string, with each RLE JSON-encoded and separated by a semicolon.\n#     \"\"\"\n#     return ';'.join([json.dumps(_rle_encode_jit(x, fg_val)) for x in masks])\n\n\n# @numba.njit\n# def _rle_decode_jit(mask_rle: npt.NDArray, height: int, width: int) -> npt.NDArray:\n#     \"\"\"\n#     s: numpy array of run-length encoding pairs (start, length)\n#     shape: (height, width) of array to return\n#     Returns numpy array, 1 - mask, 0 - background\n#     \"\"\"\n#     if len(mask_rle) % 2 != 0:\n#         # Numba requires raising a standard exception.\n#         raise ValueError('One or more rows has an odd number of values.')\n\n#     starts, lengths = mask_rle[0::2], mask_rle[1::2]\n#     starts -= 1\n#     ends = starts + lengths\n#     for i in range(len(starts) - 1):\n#         if ends[i] > starts[i + 1]:\n#             raise ValueError('Pixels must not be overlapping.')\n#     img = np.zeros(height * width, dtype=np.bool_)\n#     for lo, hi in zip(starts, ends):\n#         img[lo:hi] = 1\n#     return img\n\n\n# def rle_decode(mask_rle: str, shape: tuple[int, int]) -> npt.NDArray:\n#     \"\"\"\n#     mask_rle: run-length as string formatted (start length)\n#               empty predictions need to be encoded with '-'\n#     shape: (height, width) of array to return\n#     Returns numpy array, 1 - mask, 0 - background\n#     \"\"\"\n\n#     mask_rle = json.loads(mask_rle)\n#     mask_rle = np.asarray(mask_rle, dtype=np.int32)\n#     starts = mask_rle[0::2]\n#     if sorted(starts) != list(starts):\n#         raise ParticipantVisibleError('Submitted values must be in ascending order.')\n#     try:\n#         return _rle_decode_jit(mask_rle, shape[0], shape[1]).reshape(shape, order='F')\n#     except ValueError as e:\n#         raise ParticipantVisibleError(str(e)) from e\n\n\n# def calculate_f1_score(pred_mask: npt.NDArray, gt_mask: npt.NDArray):\n#     pred_flat = pred_mask.flatten()\n#     gt_flat = gt_mask.flatten()\n\n#     tp = np.sum((pred_flat == 1) & (gt_flat == 1))\n#     fp = np.sum((pred_flat == 1) & (gt_flat == 0))\n#     fn = np.sum((pred_flat == 0) & (gt_flat == 1))\n\n#     precision = tp / (tp + fp) if (tp + fp) > 0 else 0\n#     recall = tp / (tp + fn) if (tp + fn) > 0 else 0\n\n#     if (precision + recall) > 0:\n#         return 2 * (precision * recall) / (precision + recall)\n#     else:\n#         return 0\n\n\n# def calculate_f1_matrix(pred_masks: list[npt.NDArray], gt_masks: list[npt.NDArray]):\n#     \"\"\"\n#     Parameters:\n#     pred_masks (np.ndarray):\n#             First dimension is the number of predicted instances.\n#             Each instance is a binary mask of shape (height, width).\n#     gt_masks (np.ndarray):\n#             First dimension is the number of ground truth instances.\n#             Each instance is a binary mask of shape (height, width).\n#     \"\"\"\n\n#     num_instances_pred = len(pred_masks)\n#     num_instances_gt = len(gt_masks)\n#     f1_matrix = np.zeros((num_instances_pred, num_instances_gt))\n\n#     # Calculate F1 scores for each pair of predicted and ground truth masks\n#     for i in range(num_instances_pred):\n#         for j in range(num_instances_gt):\n#             pred_flat = pred_masks[i].flatten()\n#             gt_flat = gt_masks[j].flatten()\n#             f1_matrix[i, j] = calculate_f1_score(pred_mask=pred_flat, gt_mask=gt_flat)\n\n#     if f1_matrix.shape[0] < len(gt_masks):\n#         # Add a row of zeros to the matrix if the number of predicted instances is less than ground truth instances\n#         f1_matrix = np.vstack((f1_matrix, np.zeros((len(gt_masks) - len(f1_matrix), num_instances_gt))))\n\n#     return f1_matrix\n\n\n# def oF1_score(pred_masks: list[npt.NDArray], gt_masks: list[npt.NDArray]):\n#     \"\"\"\n#     Calculate the optimal F1 score for a set of predicted masks against\n#     ground truth masks which considers the optimal F1 score matching.\n#     This function uses the Hungarian algorithm to find the optimal assignment\n#     of predicted masks to ground truth masks based on the F1 score matrix.\n#     If the number of predicted masks is less than the number of ground truth masks,\n#     it will add a row of zeros to the F1 score matrix to ensure that the dimensions match.\n\n#     Parameters:\n#     pred_masks (list of np.ndarray): List of predicted binary masks.\n#     gt_masks (np.ndarray): Array of ground truth binary masks.\n#     Returns:\n#     float: Optimal F1 score.\n#     \"\"\"\n#     f1_matrix = calculate_f1_matrix(pred_masks, gt_masks)\n\n#     # Find the best matching between predicted and ground truth masks\n#     row_ind, col_ind = scipy.optimize.linear_sum_assignment(-f1_matrix)\n#     # The linear_sum_assignment discards excess predictions so we need a separate penalty.\n#     excess_predictions_penalty = len(gt_masks) / max(len(pred_masks), len(gt_masks))\n#     return np.mean(f1_matrix[row_ind, col_ind]) * excess_predictions_penalty\n\n\n# def evaluate_single_image(label_rles: str, prediction_rles: str, shape_str: str) -> float:\n#     shape = json.loads(shape_str)\n#     label_rles = [rle_decode(x, shape=shape) for x in label_rles.split(';')]\n#     prediction_rles = [rle_decode(x, shape=shape) for x in prediction_rles.split(';')]\n#     return oF1_score(prediction_rles, label_rles)\n\n\n# def score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n#     \"\"\"\n#     Args:\n#         solution (pd.DataFrame): The ground truth DataFrame.\n#         submission (pd.DataFrame): The submission DataFrame.\n#         row_id_column_name (str): The name of the column containing row IDs.\n#     Returns:\n#         float\n\n#     Examples\n#     --------\n#     >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['authentic', 'authentic', 'authentic'], 'shape': ['authentic', 'authentic', 'authentic']})\n#     >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['authentic', 'authentic', 'authentic']})\n#     >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n#     1.0\n\n#     >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['authentic', 'authentic', 'authentic'], 'shape': ['authentic', 'authentic', 'authentic']})\n#     >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n#     >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n#     0.0\n\n#     >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]'], 'shape': ['[720, 960]', '[720, 960]', '[720, 960]']})\n#     >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n#     >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n#     1.0\n\n#     >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 103]', '[101, 102]', '[101, 102]'], 'shape': ['[720, 960]', '[720, 960]', '[720, 960]']})\n#     >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n#     >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n#     0.9983739837398374\n\n#     >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102];[300, 100]', '[101, 102]', '[101, 102]'], 'shape': ['[720, 960]', '[720, 960]', '[720, 960]']})\n#     >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n#     >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n#     0.8333333333333334\n#     \"\"\"\n#     df = solution\n#     df = df.rename(columns={'annotation': 'label'})\n\n#     df['prediction'] = submission['annotation']\n#     # Check for correct 'authentic' label\n#     authentic_indices = (df['label'] == 'authentic') | (df['prediction'] == 'authentic')\n#     df['image_score'] = ((df['label'] == df['prediction']) & authentic_indices).astype(float)\n\n#     df.loc[~authentic_indices, 'image_score'] = df.loc[~authentic_indices].apply(\n#         lambda row: evaluate_single_image(row['label'], row['prediction'], row['shape']), axis=1\n#     )\n#     return float(np.mean(df['image_score']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.426670Z","iopub.execute_input":"2026-01-08T13:07:17.428267Z","iopub.status.idle":"2026-01-08T13:07:17.440145Z","shell.execute_reply.started":"2026-01-08T13:07:17.428220Z","shell.execute_reply":"2026-01-08T13:07:17.437698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport numpy as np\n\ndef rle_encode(mask: np.ndarray) -> str:\n    \"\"\"\n    Competition-compliant RLE encoder\n    \"\"\"\n    pixels = mask.T.flatten()  # column-major\n    dots = np.where(pixels == 1)[0]\n\n    if len(dots) < 400:\n        return \"authentic\"\n\n    runs = []\n    prev = -2\n    for i in dots:\n        if i > prev + 1:\n            runs.extend([i + 1, 0])\n        runs[-1] += 1\n        prev = i\n\n    return json.dumps([int(x) for x in runs])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.442692Z","iopub.execute_input":"2026-01-08T13:07:17.444588Z","iopub.status.idle":"2026-01-08T13:07:17.480690Z","shell.execute_reply.started":"2026-01-08T13:07:17.444537Z","shell.execute_reply":"2026-01-08T13:07:17.478784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adaptive_mask(prob):\n    mu = prob.mean()\n    sigma = prob.std()\n    thr = mu + 0.25 * sigma\n    # print(thr)\n    return (prob > thr).astype(np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.482472Z","iopub.execute_input":"2026-01-08T13:07:17.484135Z","iopub.status.idle":"2026-01-08T13:07:17.515382Z","shell.execute_reply.started":"2026-01-08T13:07:17.484086Z","shell.execute_reply":"2026-01-08T13:07:17.513392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\n\ndef clean_rle_string(rle_string):\n    cleaned = []\n    for part in rle_string.split(';'):\n        if part.strip() == '[]':\n            continue\n        cleaned.append(part)\n    return ';'.join(cleaned)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.519513Z","iopub.execute_input":"2026-01-08T13:07:17.519891Z","iopub.status.idle":"2026-01-08T13:07:17.549131Z","shell.execute_reply.started":"2026-01-08T13:07:17.519854Z","shell.execute_reply":"2026-01-08T13:07:17.548027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prob_to_mask(prob_map, threshold=None):\n    \"\"\"\n    prob_map: (H, W) float32\n    \"\"\"\n    prob_map = np.squeeze(prob_map)\n\n    if threshold is None:\n        threshold = prob_map.mean() + 2 * prob_map.std()\n    print(threshold)\n    mask = (prob_map > threshold).astype(np.uint8)\n    return mask\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.550491Z","iopub.execute_input":"2026-01-08T13:07:17.550769Z","iopub.status.idle":"2026-01-08T13:07:17.584734Z","shell.execute_reply.started":"2026-01-08T13:07:17.550745Z","shell.execute_reply":"2026-01-08T13:07:17.583069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\n\ndef clean_mask(mask):\n    kernel = np.ones((3, 3), np.uint8)\n    mask = cv2.morphologyEx(mask, cv2.MORPH_OPEN, kernel)\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)\n    return mask\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.586024Z","iopub.execute_input":"2026-01-08T13:07:17.586526Z","iopub.status.idle":"2026-01-08T13:07:17.616316Z","shell.execute_reply.started":"2026-01-08T13:07:17.586485Z","shell.execute_reply":"2026-01-08T13:07:17.615139Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# del x\n# del y\n# del x_train\n# del x_test\n# del y_train\n# del y_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.617484Z","iopub.execute_input":"2026-01-08T13:07:17.617849Z","iopub.status.idle":"2026-01-08T13:07:17.642027Z","shell.execute_reply.started":"2026-01-08T13:07:17.617802Z","shell.execute_reply":"2026-01-08T13:07:17.640518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images_dataset = []\n\n# 1 -> authentic image\n# 0 -> forged image\n# 2 -> masks image\n\nfor i in test_dir.iterdir():\n    images_dataset.append({\"case_id\":i.stem,\"path\":i})\n\n#j = 0\n#for i in authentic_dir.iterdir():\n#    images_dataset.append({\"path\":i,\"case_id\":i.stem})\n#    j+=1\n#    if j==10:\n#        break\n\ntest_df = pd.DataFrame(images_dataset)\n# new_df[\"label\"].value_counts()\ntest_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.643174Z","iopub.execute_input":"2026-01-08T13:07:17.643450Z","iopub.status.idle":"2026-01-08T13:07:17.683206Z","shell.execute_reply.started":"2026-01-08T13:07:17.643424Z","shell.execute_reply":"2026-01-08T13:07:17.682078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.684300Z","iopub.execute_input":"2026-01-08T13:07:17.684643Z","iopub.status.idle":"2026-01-08T13:07:17.712579Z","shell.execute_reply.started":"2026-01-08T13:07:17.684609Z","shell.execute_reply":"2026-01-08T13:07:17.711710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from PIL import Image\n# import numpy as np\n# from pathlib import Path\n\n# submission = []\n\n# for i in test_df[\"path\"]:\n#     file = Path(i)\n#     # ---- Load image ----\n#     img = Image.open(file).convert(\"RGB\")\n#     img = img.resize((256, 256))\n#     img = np.array(img, dtype=np.float32) / 255.0\n#     img = np.expand_dims(img, axis=0)\n\n#     mask = model.predict(img)[0]\n#     mask = prob_to_mask(mask,0.30)\n#     mask = clean_mask(mask)\n\n#     name = file.stem\n\n#     if mask.sum() < 300:\n#         submission.append({\"case_id\":name,\"annotation\":\"authentic\"})\n#     else:\n#         ans = rle_encode([mask])\n#         # ans2 = rle_encode([mask])\n#         # print(type(ans2))\n#         # print(ans2)\n#         ans = json.dumps(ans)\n#         submission.append({\"case_id\":name,\"annotation\":str(ans)})\n\n# submission = pd.DataFrame(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.713860Z","iopub.execute_input":"2026-01-08T13:07:17.714290Z","iopub.status.idle":"2026-01-08T13:07:17.721980Z","shell.execute_reply.started":"2026-01-08T13:07:17.714257Z","shell.execute_reply":"2026-01-08T13:07:17.720994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.723271Z","iopub.execute_input":"2026-01-08T13:07:17.724031Z","iopub.status.idle":"2026-01-08T13:07:17.749273Z","shell.execute_reply.started":"2026-01-08T13:07:17.723897Z","shell.execute_reply":"2026-01-08T13:07:17.747694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# assert len(submission) == len(test_df)\n# assert submission['annotation'].notnull().all()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.750465Z","iopub.execute_input":"2026-01-08T13:07:17.750842Z","iopub.status.idle":"2026-01-08T13:07:17.771286Z","shell.execute_reply.started":"2026-01-08T13:07:17.750805Z","shell.execute_reply":"2026-01-08T13:07:17.770077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# score(solution,submission,\"case_id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.772673Z","iopub.execute_input":"2026-01-08T13:07:17.773134Z","iopub.status.idle":"2026-01-08T13:07:17.795588Z","shell.execute_reply.started":"2026-01-08T13:07:17.773046Z","shell.execute_reply":"2026-01-08T13:07:17.794631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Sample rows     :\", len(test_df))\n# print(\"Submission rows :\", len(submission))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.797145Z","iopub.execute_input":"2026-01-08T13:07:17.797411Z","iopub.status.idle":"2026-01-08T13:07:17.818883Z","shell.execute_reply.started":"2026-01-08T13:07:17.797386Z","shell.execute_reply":"2026-01-08T13:07:17.817697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission = (\n#     test_df.merge(submission, on=\"case_id\", how=\"left\", suffixes=(\"\", \"_pred\"))\n#     .assign(annotation=lambda x: x[\"annotation\"].fillna(\"authentic\"))\n#     [[\"case_id\", \"annotation\"]]\n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.819962Z","iopub.execute_input":"2026-01-08T13:07:17.820308Z","iopub.status.idle":"2026-01-08T13:07:17.852639Z","shell.execute_reply.started":"2026-01-08T13:07:17.820276Z","shell.execute_reply":"2026-01-08T13:07:17.851433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission[\"case_id\"] = submission[\"case_id\"] .astype(\"int64\")\n# submission.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.854016Z","iopub.execute_input":"2026-01-08T13:07:17.854395Z","iopub.status.idle":"2026-01-08T13:07:17.878101Z","shell.execute_reply.started":"2026-01-08T13:07:17.854356Z","shell.execute_reply":"2026-01-08T13:07:17.876947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.879446Z","iopub.execute_input":"2026-01-08T13:07:17.880475Z","iopub.status.idle":"2026-01-08T13:07:17.909885Z","shell.execute_reply.started":"2026-01-08T13:07:17.880432Z","shell.execute_reply":"2026-01-08T13:07:17.908548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.911380Z","iopub.execute_input":"2026-01-08T13:07:17.911738Z","iopub.status.idle":"2026-01-08T13:07:17.938241Z","shell.execute_reply.started":"2026-01-08T13:07:17.911703Z","shell.execute_reply":"2026-01-08T13:07:17.937078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nfrom pathlib import Path\n\nsubmission = []\n\nfor i in test_df[\"path\"]:\n    file = Path(i)\n    pil = Image.open(file).convert(\"RGB\")\n    orig_w, orig_h = pil.size\n\n    img = pil.resize((256, 256))\n    img = np.array(img, dtype=np.float32) / 255.0\n    img = np.expand_dims(img, axis=0)\n\n    prob = model.predict(img)[0]\n    prob = cv2.resize(prob, (orig_w, orig_h))\n\n    mask = prob_to_mask(prob,1)\n    # mask = adaptive_mask(prob)\n    mask = clean_mask(mask)\n    mask = (mask > 0).astype(np.uint8) \n    # print(mask)\n\n    case_id = file.stem\n\n    # if mask.sum() < 300:\n    #     annotation = \"authentic\"\n    # else:\n    annotation = rle_encode(mask)\n\n    submission.append({\n        \"case_id\": str(case_id),\n        \"annotation\": annotation\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:17.939287Z","iopub.execute_input":"2026-01-08T13:07:17.939621Z","iopub.status.idle":"2026-01-08T13:07:21.137721Z","shell.execute_reply.started":"2026-01-08T13:07:17.939585Z","shell.execute_reply":"2026-01-08T13:07:21.136706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame(submission)\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.138941Z","iopub.execute_input":"2026-01-08T13:07:21.139640Z","iopub.status.idle":"2026-01-08T13:07:21.149142Z","shell.execute_reply.started":"2026-01-08T13:07:21.139598Z","shell.execute_reply":"2026-01-08T13:07:21.147621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = pd.read_csv(\n    \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv\",\n    dtype={\"case_id\": str}\n)\nsubmission[\"case_id\"] = submission[\"case_id\"].astype(str)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.150315Z","iopub.execute_input":"2026-01-08T13:07:21.150690Z","iopub.status.idle":"2026-01-08T13:07:21.185582Z","shell.execute_reply.started":"2026-01-08T13:07:21.150664Z","shell.execute_reply":"2026-01-08T13:07:21.184676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final = sample[[\"case_id\"]].merge(\n    submission,\n    on=\"case_id\",\n    how=\"left\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.186986Z","iopub.execute_input":"2026-01-08T13:07:21.187320Z","iopub.status.idle":"2026-01-08T13:07:21.211121Z","shell.execute_reply.started":"2026-01-08T13:07:21.187288Z","shell.execute_reply":"2026-01-08T13:07:21.210226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final[\"annotation\"] = final[\"annotation\"].fillna(\"authentic\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.212184Z","iopub.execute_input":"2026-01-08T13:07:21.212437Z","iopub.status.idle":"2026-01-08T13:07:21.231264Z","shell.execute_reply.started":"2026-01-08T13:07:21.212415Z","shell.execute_reply":"2026-01-08T13:07:21.230286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.232542Z","iopub.execute_input":"2026-01-08T13:07:21.232890Z","iopub.status.idle":"2026-01-08T13:07:21.260631Z","shell.execute_reply.started":"2026-01-08T13:07:21.232852Z","shell.execute_reply":"2026-01-08T13:07:21.259688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.261770Z","iopub.execute_input":"2026-01-08T13:07:21.262136Z","iopub.status.idle":"2026-01-08T13:07:21.290332Z","shell.execute_reply.started":"2026-01-08T13:07:21.262082Z","shell.execute_reply":"2026-01-08T13:07:21.289415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"submission.csv\")\nprint(df.head())\nprint(df.dtypes)\nprint(df[\"annotation\"].apply(type).value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T13:07:21.291468Z","iopub.execute_input":"2026-01-08T13:07:21.291786Z","iopub.status.idle":"2026-01-08T13:07:21.318454Z","shell.execute_reply.started":"2026-01-08T13:07:21.291760Z","shell.execute_reply":"2026-01-08T13:07:21.317402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}