{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torchvision\nfrom torch.utils.data import DataLoader\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torchvision.models.detection import MaskRCNN\nfrom torchvision.models.detection.backbone_utils import resnet_fpn_backbone\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\n# ===== CONFIGURATION =====\nIMG_SIZE = 512\nBATCH_SIZE = 4\nNUM_EPOCHS = 15\nLEARNING_RATE = 1e-4\nTHRESHOLD = 0.4\nNUM_CLASSES = 2\n","metadata":{"_uuid":"23768d05-ddd2-40a4-87c8-6a50af4833df","_cell_guid":"8f30142e-b853-411f-ab4f-0528809fe2a2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-07T05:18:04.882994Z","iopub.execute_input":"2025-11-07T05:18:04.883683Z","iopub.status.idle":"2025-11-07T05:18:04.888484Z","shell.execute_reply.started":"2025-11-07T05:18:04.883653Z","shell.execute_reply":"2025-11-07T05:18:04.887529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===== AUGMENTATIONS =====\ntrain_transform = A.Compose([\n    A.HorizontalFlip(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.RandomRotate90(p=0.5),\n    A.RandomBrightnessContrast(p=0.2),\n    A.Normalize(mean=(0.485,0.456,0.406), std=(0.229,0.224,0.225)),\n    ToTensorV2()\n])\n\n# ===== CUSTOM DATASET =====\nclass CustomDataset(torch.utils.data.Dataset):\n    def __init__(self, image_paths, mask_paths, transform=None):\n        self.image_paths = image_paths\n        self.mask_paths = mask_paths\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, idx):\n        image = ... # Load image from self.image_paths[idx]\n        mask = ...  # Load mask from self.mask_paths[idx]\n\n        if self.transform:\n            augmented = self.transform(image=image, mask=mask)\n            image = augmented['image']\n            mask = augmented['mask']\n\n        target = {\n            'boxes': torch.tensor([[0,0,mask.shape[1],mask.shape[0]]], dtype=torch.float32),\n            'labels': torch.ones((1,), dtype=torch.int64),\n            'masks': torch.tensor(mask[None], dtype=torch.uint8)\n        }\n        return image, target\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:18:07.547844Z","iopub.execute_input":"2025-11-07T05:18:07.548115Z","iopub.status.idle":"2025-11-07T05:18:07.558377Z","shell.execute_reply.started":"2025-11-07T05:18:07.548092Z","shell.execute_reply":"2025-11-07T05:18:07.557615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===== DEVICE SELECTION =====\ndevice = None\ntry:\n    import torch_xla.core.xla_model as xm\n    device = xm.xla_device()\n    print(\"Device: TPU\")\nexcept ImportError:\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    print(f\"Device: {device}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:18:10.437590Z","iopub.execute_input":"2025-11-07T05:18:10.437861Z","iopub.status.idle":"2025-11-07T05:18:10.443777Z","shell.execute_reply.started":"2025-11-07T05:18:10.437839Z","shell.execute_reply":"2025-11-07T05:18:10.443171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===== MASK R-CNN MODEL =====\ndef get_maskrcnn_model(num_classes=NUM_CLASSES):\n    backbone = resnet_fpn_backbone('resnet101', pretrained=True)\n    model = MaskRCNN(backbone, num_classes=num_classes)\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:18:14.079598Z","iopub.execute_input":"2025-11-07T05:18:14.079868Z","iopub.status.idle":"2025-11-07T05:18:14.084332Z","shell.execute_reply.started":"2025-11-07T05:18:14.079847Z","shell.execute_reply":"2025-11-07T05:18:14.083542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===== TRAINING FUNCTION =====\ndef train_one_epoch(model, dataloader, optimizer, device):\n    model.train()\n    total_loss = 0\n    for images, targets in tqdm(dataloader):\n        images = list(img.to(device) for img in images)\n        targets = [{k:v.to(device) for k,v in t.items()} for t in targets]\n\n        loss_dict = model(images, targets)\n        losses = sum(loss for loss in loss_dict.values())\n        optimizer.zero_grad()\n        losses.backward()\n        optimizer.step()\n        total_loss += losses.item()\n    return total_loss / len(dataloader)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:18:16.605060Z","iopub.execute_input":"2025-11-07T05:18:16.605369Z","iopub.status.idle":"2025-11-07T05:18:16.610748Z","shell.execute_reply.started":"2025-11-07T05:18:16.605345Z","shell.execute_reply":"2025-11-07T05:18:16.609925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===== PREDICTION FUNCTION =====\ndef predict(model, image_paths, device, img_size=IMG_SIZE, threshold=THRESHOLD):\n    model.eval()\n    predictions = {}\n    for img_path in image_paths:\n        image = ... # Load and resize to img_size\n        with torch.no_grad():\n            output = model([image.to(device)])[0]\n\n        # Remove tiny masks\n        masks = output['masks'].cpu().numpy() > threshold\n        keep_masks = [m for m in masks if m.sum() > 100]\n\n        predictions[img_path] = \"forged\" if len(keep_masks)>0 else \"authentic\"\n    return predictions\n\n# ===== SUBMISSION CREATION =====\ndef create_submission(predictions, sample_csv):\n    sample = pd.read_csv(sample_csv)\n    submission_data = []\n    for _, row in sample.iterrows():\n        case_id = row['case_id']\n        annotation = predictions.get(case_id, \"authentic\")\n        submission_data.append({'case_id': case_id, 'annotation': annotation})\n    submission_df = pd.DataFrame(submission_data)\n    submission_df.to_csv('submission.csv', index=False)\n    print(\"Submission saved!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:18:19.211391Z","iopub.execute_input":"2025-11-07T05:18:19.211709Z","iopub.status.idle":"2025-11-07T05:18:19.218454Z","shell.execute_reply.started":"2025-11-07T05:18:19.211680Z","shell.execute_reply":"2025-11-07T05:18:19.217800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Define paths at a module level\npaths = {\n    'synthetic_forged': '/path/to/forged_images',\n    'synthetic_masks': '/path/to/masks'\n}\n\n# Define your CustomDataset and transforms here\n# from your_module import CustomDataset, train_transform\n\ndef main(image_paths_dict):\n    # DataLoader\n    train_dataset = CustomDataset(\n        image_paths=image_paths_dict['synthetic_forged'],\n        mask_paths=image_paths_dict['synthetic_masks'],\n        transform=train_transform\n    )\n    # ... rest of your main logic ...\n\nif __name__ == \"__main__\":\n    # Pass the paths dictionary when calling main\n    main(paths)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:19:10.263707Z","iopub.execute_input":"2025-11-07T05:19:10.264425Z","iopub.status.idle":"2025-11-07T05:19:10.268522Z","shell.execute_reply.started":"2025-11-07T05:19:10.264400Z","shell.execute_reply":"2025-11-07T05:19:10.267728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define your CustomDataset and transforms here\n# from your_module import CustomDataset, train_transform\n\ndef main():\n    # Define paths inside main\n    paths = {\n        'synthetic_forged': '/path/to/forged_images',\n        'synthetic_masks': '/path/to/masks'\n    }\n    \n    # DataLoader\n    train_dataset = CustomDataset(\n        image_paths=paths['synthetic_forged'],\n        mask_paths=paths['synthetic_masks'],\n        transform=train_transform\n    )\n    # ... rest of your main logic ...\n\nif __name__ == \"__main__\":\n    main()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:19:23.142393Z","iopub.execute_input":"2025-11-07T05:19:23.143084Z","iopub.status.idle":"2025-11-07T05:19:23.147484Z","shell.execute_reply.started":"2025-11-07T05:19:23.143051Z","shell.execute_reply":"2025-11-07T05:19:23.146792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define paths at a module level\npaths = {\n    'synthetic_forged': '/path/to/forged_images',\n    'synthetic_masks': '/path/to/masks'\n}\n\n# Define your CustomDataset and transforms here\n# from your_module import CustomDataset, train_transform\n\ndef main():\n    # Explicitly state that you are using the global paths variable\n    global paths \n\n    # DataLoader\n    train_dataset = CustomDataset(\n        image_paths=paths['synthetic_forged'],\n        mask_paths=paths['synthetic_masks'],\n        transform=train_transform\n    )\n    # ... rest of your main logic ...\n\nif __name__ == \"__main__\":\n    main()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:19:43.840049Z","iopub.execute_input":"2025-11-07T05:19:43.840339Z","iopub.status.idle":"2025-11-07T05:19:43.844740Z","shell.execute_reply.started":"2025-11-07T05:19:43.840316Z","shell.execute_reply":"2025-11-07T05:19:43.843985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\n# You would need to define your specific transforms here (e.g., using torchvision.transforms)\n# from torchvision import transforms \n\n# --- Dummy CustomDataset and transforms for demonstration ---\n# Replace this with your actual implementation\nclass CustomDataset(Dataset):\n    def __init__(self, image_paths, mask_paths, transform=None):\n        # In a real scenario, you'd list all image files in these directories\n        self.image_paths = [os.path.join(image_paths, f) for f in os.listdir(image_paths)]\n        self.mask_paths = [os.path.join(mask_paths, f) for f in os.listdir(mask_paths)]\n        self.transform = transform\n        print(f\"Initialized CustomDataset with {len(self.image_paths)} images and {len(self.mask_paths)} masks.\")\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, idx):\n        # Dummy implementation: In reality, load and process actual image/mask files\n        image = torch.randn(3, 256, 256) \n        mask = torch.randn(1, 256, 256)\n        if self.transform:\n            image = self.transform(image)\n        return image, mask\n\n# Dummy transforms\ntrain_transform = lambda x: x # Identity transform for this example\n# ----------------------------------------------------------------\n\ndef main(paths):\n    \"\"\"\n    Main function that now accepts the paths dictionary as an argument.\n    \"\"\"\n    print(\"Starting main function...\")\n    print(f\"Forged path: {paths['synthetic_forged']}\")\n    print(f\"Mask path: {paths['synthetic_masks']}\")\n    \n    # DataLoader\n    train_dataset = CustomDataset(\n        image_paths=paths['synthetic_forged'],\n        mask_paths=paths['synthetic_masks'],\n        transform=train_transform\n    )\n    \n    # Example usage:\n    # train_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\n\nif __name__ == \"__main__\":\n    # 1. Define the base input directory path\n    BASE_DIR = '/kaggle/input/recodai-luc-scientific-image-forgery-detection'\n\n    # 2. Define the 'paths' dictionary with the correct Kaggle paths\n    paths = {\n        # These are assumed subdirectories within the competition data\n        'synthetic_forged': os.path.join(BASE_DIR, 'train_images', 'forged'), \n        'synthetic_masks': os.path.join(BASE_DIR, 'train_masks')\n    }\n    \n    # 3. Call main() and pass the 'paths' dictionary\n    main(paths)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:21:27.038622Z","iopub.execute_input":"2025-11-07T05:21:27.039207Z","iopub.status.idle":"2025-11-07T05:21:27.074252Z","shell.execute_reply.started":"2025-11-07T05:21:27.039174Z","shell.execute_reply":"2025-11-07T05:21:27.073557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nfrom torch.utils.data import Dataset\nfrom PIL import Image\nimport numpy as np\n\nclass ForgeryDetectionDataset(Dataset):\n    def __init__(self, image_dir, mask_dir, transform=None):\n        self.image_dir = image_dir\n        self.mask_dir = mask_dir\n        self.transform = transform\n        # Assumes image filenames and mask filenames match exactly (e.g., 'img_1.png' and 'img_1.png')\n        self.image_filenames = sorted(os.listdir(image_dir))\n        self.mask_filenames = sorted(os.listdir(mask_dir))\n\n    def __len__(self):\n        return len(self.image_filenames)\n\n    def __getitem__(self, idx):\n        img_name = self.image_filenames[idx]\n        mask_name = self.mask_filenames[idx] # Ensure alignment is correct\n\n        img_path = os.path.join(self.image_dir, img_name)\n        mask_path = os.path.join(self.mask_dir, mask_name)\n\n        # Load image (RGB)\n        image = Image.open(img_path).convert(\"RGB\")\n        # Load mask (Grayscale/Binary) and ensure it's a binary tensor (0 or 1)\n        mask = Image.open(mask_path).convert(\"L\") \n        mask = np.array(mask) > 0 # Convert to boolean/binary array\n        mask = mask.astype(np.float32) # Convert to float32 for PyTorch loss functions\n        \n        # Apply transformations if provided\n        if self.transform:\n            # Note: A real implementation needs synchronized transforms for img and mask\n            image = self.transform(image)\n            # You might need specific mask transforms if you resize/crop randomly\n            # mask = transform_mask(mask) \n\n        # Convert numpy array to tensor (add channel dimension for mask)\n        image = torch.from_numpy(np.array(image)).permute(2, 0, 1) # HWC to CHW for image\n        mask = torch.from_numpy(mask).unsqueeze(0) # Add channel dimension for mask\n\n        return image, mask\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:22:13.161439Z","iopub.execute_input":"2025-11-07T05:22:13.162090Z","iopub.status.idle":"2025-11-07T05:22:13.168436Z","shell.execute_reply.started":"2025-11-07T05:22:13.162066Z","shell.execute_reply":"2025-11-07T05:22:13.167717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This is a placeholder. You usually import a pre-built U-Net library \n# or copy the U-Net architecture definition here.\n\nimport torch\nimport torch.nn as nn\n# A full U-Net implementation is too long for this response, \n# but you can use a library like 'segmentation_models_pytorch' (smp) in Kaggle:\n\n# In your notebook: !pip install segmentation-models-pytorch\n# import segmentation_models_pytorch as smp\n\n# def get_model():\n#     model = smp.Unet(\n#         encoder_name=\"resnet34\",      # Choose encoder\n#         encoder_weights=\"imagenet\",   # Use pre-trained weights\n#         in_channels=3,                # RGB images\n#         classes=1,                    # Binary segmentation (forgery/not forgery)\n#     )\n#     return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:22:30.318535Z","iopub.execute_input":"2025-11-07T05:22:30.318990Z","iopub.status.idle":"2025-11-07T05:22:30.322800Z","shell.execute_reply.started":"2025-11-07T05:22:30.318968Z","shell.execute_reply":"2025-11-07T05:22:30.322224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\ndef iou_score(outputs, labels):\n    # Apply sigmoid and threshold to get binary predictions (0 or 1)\n    outputs = torch.sigmoid(outputs)\n    outputs = (outputs > 0.5).float()\n    \n    intersection = (outputs * labels).sum()\n    union = (outputs + labels).sum() - intersection\n    \n    # Handle the case where both prediction and label are entirely empty\n    iou = (intersection + 1e-6) / (union + 1e-6) \n    return iou\n\n# F1 Score is mathematically identical to the Dice coefficient for binary segmentation\ndef f1_score(outputs, labels):\n    # Same thresholding as IoU\n    outputs = torch.sigmoid(outputs)\n    outputs = (outputs > 0.5).float()\n    \n    # Dice coefficient calculation\n    intersection = (outputs * labels).sum()\n    dice = (2. * intersection + 1e-6) / (outputs.sum() + labels.sum() + 1e-6)\n    return dice\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:22:38.717911Z","iopub.execute_input":"2025-11-07T05:22:38.718169Z","iopub.status.idle":"2025-11-07T05:22:38.723550Z","shell.execute_reply.started":"2025-11-07T05:22:38.718150Z","shell.execute_reply":"2025-11-07T05:22:38.722819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nfrom torch.utils.data import Dataset\nimport numpy as np\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2\n\nclass ForgeryDetectionDataset(Dataset):\n    def __init__(self, image_dir, mask_dir, transform=None):\n        self.image_dir = image_dir\n        self.mask_dir = mask_dir\n        self.transform = transform\n        self.image_filenames = sorted(os.listdir(image_dir))\n        self.mask_filenames = sorted(os.listdir(mask_dir))\n\n    def __len__(self):\n        return len(self.image_filenames)\n\n    def __getitem__(self, idx):\n        img_name = self.image_filenames[idx]\n        mask_name = self.mask_filenames[idx] \n        img_path = os.path.join(self.image_dir, img_name)\n        mask_path = os.path.join(self.mask_dir, mask_name)\n\n        # Load image (PIL -> NumPy array HWC, RGB)\n        image = np.array(Image.open(img_path).convert(\"RGB\"))\n        \n        # Load mask (.npy file)\n        mask = np.load(mask_path).astype(np.float32)\n        # Ensure mask is binary if necessary\n        if mask.max() > 1:\n            mask = (mask > 0).astype(np.float32)\n\n        # Apply transformations if provided (Albumentations handles image and mask together)\n        if self.transform:\n            augmented = self.transform(image=image, mask=mask)\n            image = augmented['image']\n            mask = augmented['mask']\n        \n        # Ensure mask has a channel dimension [1, H, W] for PyTorch compatibility\n        if mask.ndim == 2:\n            mask = mask[None, :, :] # Adds a channel dimension\n\n        return image, mask\n\n# Example of standard transformations we will use in main.py:\ndef get_train_transforms(image_size=256):\n    return A.Compose([\n        A.Resize(height=image_size, width=image_size),\n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ToTensorV2(), # Converts HWC to CHW and numpy to tensor\n    ], p=1.0)\n    \ndef get_val_transforms(image_size=256):\n    return A.Compose([\n        A.Resize(height=image_size, width=image_size),\n        A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ToTensorV2(),\n    ], p=1.0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:24:42.230553Z","iopub.execute_input":"2025-11-07T05:24:42.230816Z","iopub.status.idle":"2025-11-07T05:24:42.239748Z","shell.execute_reply.started":"2025-11-07T05:24:42.230797Z","shell.execute_reply":"2025-11-07T05:24:42.239006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import DataLoader\nfrom PIL import Image\n# Import your dataset class, model, and transforms here\n\n# --- RLE Helper Function (Standard for Kaggle Segments) ---\ndef rle_encode(mask):\n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n# -----------------------------------------------------------\n\ndef generate_submission(model_path, test_image_dir, submission_csv_path, device):\n    # Load the best model weights\n    model = get_unet_model().to(device)\n    model.load_state_dict(torch.load(model_path, map_location=device))\n    model.eval()\n\n    # Create a test dataset/loader (reuse ForgeryDetectionDataset structure, but no masks needed)\n    # We need a simplified test dataset that just loads images and their filenames\n    class TestDataset(Dataset):\n        def __init__(self, image_dir, transform=None):\n            self.image_dir = image_dir\n            self.image_filenames = sorted(os.listdir(image_dir))\n            self.transform = transform\n        def __len__(self): return len(self.image_filenames)\n        def __getitem__(self, idx):\n            img_name = self.image_filenames[idx]\n            img_path = os.path.join(self.image_dir, img_name)\n            image = np.array(Image.open(img_path).convert(\"RGB\"))\n            if self.transform:\n                augmented = self.transform(image=image) # Only transform image\n                image = augmented['image']\n            return image, img_name # Return image tensor and filename\n\n    test_transforms = get_val_transforms() # Use validation transforms\n    test_dataset = TestDataset(test_image_dir, transform=test_transforms)\n    test_loader = DataLoader(test_dataset, batch_size=BATCH_SIZE, shuffle=False)\n    \n    submission_list = []\n    \n    with torch.no_grad():\n        for images, names in test_loader:\n            images = images.to(device)\n            outputs = model(images)\n            # Apply sigmoid and threshold to get binary mask [B, 1, H, W]\n            preds = torch.sigmoid(outputs)\n            preds = (preds > 0.5).cpu().numpy().astype(np.uint8) \n\n            # Process batch predictions one by one\n            for pred_mask, name in zip(preds, names):\n                pred_mask = pred_mask.squeeze() # Remove channel dimension [H, W]\n                rle = rle_encode(pred_mask)\n                submission_list.append({'id': name, 'predicted': rle})\n\n    # Create the final pandas DataFrame and save to CSV\n    submission_df = pd.DataFrame(submission_list)\n    submission_df.to_csv(submission_csv_path, index=False)\n    print(f\"Successfully generated submission file: {submission_csv_path}\")\n\n\n# --- How to run the submission generation in your main notebook ---\nif __name__ == \"__main__\":\n    # Ensure you replace 'best_model.pth' with the path you saved your actual model to\n    # and adjust the TEST_IMG_DIR path for the competition structure\n    TEST_IMG_DIR = os.path.join('/kaggle/input/recodai-luc-scientific-image-forgery-detection', 'test_images')\n    SUBMISSION_FILE_PATH = 'submission.csv'\n    \n    # You would typically run the training first, save the model, then run this:\n    # generate_submission(\n    #     model_path='best_model.pth', \n    #     test_image_dir=TEST_IMG_DIR, \n    #     submission_csv_path=SUBMISSION_FILE_PATH, \n    #     device=torch.device(\"cuda\")\n    # )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:25:07.029821Z","iopub.execute_input":"2025-11-07T05:25:07.030447Z","iopub.status.idle":"2025-11-07T05:25:07.040801Z","shell.execute_reply.started":"2025-11-07T05:25:07.030422Z","shell.execute_reply":"2025-11-07T05:25:07.039938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nComplete Kaggle Notebook: Synthetic Data Generation + Mask R-CNN Training\nRuns entirely on Kaggle GPU - no data uploads required\n\"\"\"\n\nimport os\nimport cv2\nimport json\nimport time\nimport torch\nimport random\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision.models.detection import maskrcnn_resnet50_fpn_v2\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection.mask_rcnn import MaskRCNNPredictor\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# ============================================================\n# PART 1: SYNTHETIC DATA GENERATION\n# ============================================================\n\ndef generate_synthetic_forgeries(authentic_path, forged_path, masks_path,\n                                 output_forged_path, output_masks_path,\n                                 num_synthetic=5000):\n    \"\"\"Generate synthetic forgeries using Poisson blending\"\"\"\n\n    print(\"\\n\" + \"=\"*60)\n    print(\"SYNTHETIC FORGERY GENERATION\")\n    print(\"=\"*60)\n\n    os.makedirs(output_forged_path, exist_ok=True)\n    os.makedirs(output_masks_path, exist_ok=True)\n\n    # Load source images\n    authentic_files = [f for f in os.listdir(authentic_path)\n                      if f.lower().endswith(('.png', '.jpg', '.jpeg'))]\n    forged_files = [f for f in os.listdir(forged_path)\n                   if f.lower().endswith(('.png', '.jpg', '.jpeg'))]\n\n    print(f\"Source data: {len(authentic_files)} authentic, {len(forged_files)} forged\")\n    print(f\"Generating {num_synthetic} synthetic forgeries...\")\n\n    generated = 0\n    attempts = 0\n    max_attempts = num_synthetic * 3\n\n    with tqdm(total=num_synthetic, desc=\"Generating\") as pbar:\n        while generated < num_synthetic and attempts < max_attempts:\n            attempts += 1\n\n            try:\n                # Random source and target\n                source_file = random.choice(forged_files)\n                target_file = random.choice(authentic_files)\n\n                # Load images\n                source_img_path = os.path.join(forged_path, source_file)\n                target_img_path = os.path.join(authentic_path, target_file)\n\n                source_img = cv2.imread(source_img_path)\n                target_img = cv2.imread(target_img_path)\n\n                if source_img is None or target_img is None:\n                    continue\n\n                # Load source mask\n                mask_file = f\"{source_file.split('.')[0]}.npy\"\n                mask_path = os.path.join(masks_path, mask_file)\n\n                if not os.path.exists(mask_path):\n                    continue\n\n                source_mask = np.load(mask_path)\n                if source_mask.ndim == 3:\n                    source_mask = source_mask.max(axis=0) if source_mask.shape[0] <= 10 else source_mask.max(axis=-1)\n\n                # Resize to match target\n                h, w = target_img.shape[:2]\n                source_img = cv2.resize(source_img, (w, h))\n                source_mask = cv2.resize(source_mask.astype(np.uint8), (w, h))\n                source_mask = (source_mask > 0).astype(np.uint8) * 255\n\n                # Find contours and select random region\n                contours, _ = cv2.findContours(source_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n                if len(contours) == 0:\n                    continue\n\n                contour = random.choice(contours)\n                x, y, cw, ch = cv2.boundingRect(contour)\n\n                if cw < 20 or ch < 20:  # Skip tiny regions\n                    continue\n\n                # Create region mask\n                region_mask = np.zeros_like(source_mask)\n                cv2.drawContours(region_mask, [contour], -1, 255, -1)\n\n                # Random paste location\n                max_x = max(0, w - cw - 10)\n                max_y = max(0, h - ch - 10)\n                if max_x <= 0 or max_y <= 0:\n                    continue\n\n                paste_x = random.randint(0, max_x)\n                paste_y = random.randint(0, max_y)\n\n                # Extract region\n                region_img = source_img[y:y+ch, x:x+cw].copy()\n                region_mask_crop = region_mask[y:y+ch, x:x+cw].copy()\n\n                # Poisson blending (seamless clone)\n                center = (paste_x + cw//2, paste_y + ch//2)\n\n                try:\n                    result = cv2.seamlessClone(\n                        region_img,\n                        target_img.copy(),\n                        region_mask_crop,\n                        center,\n                        cv2.NORMAL_CLONE\n                    )\n                except:\n                    # Fallback to simple paste\n                    result = target_img.copy()\n                    result[paste_y:paste_y+ch, paste_x:paste_x+cw] = np.where(\n                        region_mask_crop[:,:,None] > 0,\n                        region_img,\n                        result[paste_y:paste_y+ch, paste_x:paste_x+cw]\n                    )\n\n                # Create output mask\n                output_mask = np.zeros((h, w), dtype=np.uint8)\n                output_mask[paste_y:paste_y+ch, paste_x:paste_x+cw] = (region_mask_crop > 0).astype(np.uint8)\n\n                # Save\n                output_file = f\"synthetic_{source_file.split('.')[0]}_{generated}.png\"\n                cv2.imwrite(os.path.join(output_forged_path, output_file), result)\n                np.save(os.path.join(output_masks_path, output_file.replace('.png', '.npy')), output_mask)\n\n                generated += 1\n                pbar.update(1)\n\n            except Exception as e:\n                continue\n\n    print(f\"\\nGenerated {generated} synthetic forgeries\")\n    return generated\n\n\n# ============================================================\n# PART 2: DATASET CLASS\n# ============================================================\n\nclass ForgeryDatasetCombined(Dataset):\n    \"\"\"Dataset combining original and synthetic forgery data\"\"\"\n\n    def __init__(self, authentic_path, forged_path, masks_path,\n                 synthetic_forged_path, synthetic_masks_path,\n                 img_size=256, max_original=None, max_synthetic=None):\n        self.img_size = img_size\n        self.samples = []\n\n        # Collect authentic samples\n        if os.path.exists(authentic_path):\n            files = sorted(os.listdir(authentic_path))\n            if max_original:\n                files = files[:max_original // 2]\n\n            for file in files:\n                if file.lower().endswith(('.png', '.jpg', '.jpeg')):\n                    img_path = os.path.join(authentic_path, file)\n                    self.samples.append((img_path, None, False))\n\n        # Collect original forged samples\n        if os.path.exists(forged_path):\n            files = sorted(os.listdir(forged_path))\n            if max_original:\n                files = files[:max_original // 2]\n\n            for file in files:\n                if file.lower().endswith(('.png', '.jpg', '.jpeg')):\n                    img_path = os.path.join(forged_path, file)\n                    mask_path = os.path.join(masks_path, f\"{file.split('.')[0]}.npy\")\n                    if os.path.exists(mask_path):\n                        self.samples.append((img_path, mask_path, True))\n\n        # Collect synthetic forged samples\n        if os.path.exists(synthetic_forged_path):\n            files = sorted(os.listdir(synthetic_forged_path))\n            if max_synthetic:\n                files = files[:max_synthetic]\n\n            for file in files:\n                if file.lower().endswith(('.png', '.jpg', '.jpeg')):\n                    img_path = os.path.join(synthetic_forged_path, file)\n                    mask_file = file.rsplit('.', 1)[0] + '.npy'\n                    mask_path = os.path.join(synthetic_masks_path, mask_file)\n                    if os.path.exists(mask_path):\n                        self.samples.append((img_path, mask_path, True))\n\n        print(f\"Loaded {len(self.samples)} samples\")\n\n    def __len__(self):\n        return len(self.samples)\n\n    def __getitem__(self, idx):\n        img_path, mask_path, is_forged = self.samples[idx]\n\n        # Load image\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, (self.img_size, self.img_size))\n        img = img.astype(np.float32) / 255.0\n        img = torch.from_numpy(img).permute(2, 0, 1)\n\n        target = {}\n\n        if is_forged and mask_path:\n            try:\n                mask = np.load(mask_path)\n                if mask.ndim == 3:\n                    mask = mask.max(axis=0) if mask.shape[0] <= 10 else mask.max(axis=-1)\n                mask = cv2.resize(mask.astype(np.uint8), (self.img_size, self.img_size))\n                mask = (mask > 0).astype(np.uint8)\n\n                contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n\n                if len(contours) > 0:\n                    masks, boxes = [], []\n                    for contour in contours:\n                        instance_mask = np.zeros_like(mask)\n                        cv2.drawContours(instance_mask, [contour], -1, 1, -1)\n                        x, y, w, h = cv2.boundingRect(contour)\n                        if w > 5 and h > 5:\n                            boxes.append([x, y, x + w, y + h])\n                            masks.append(instance_mask)\n\n                    if len(masks) > 0:\n                        target['boxes'] = torch.as_tensor(boxes, dtype=torch.float32)\n                        target['labels'] = torch.ones((len(boxes),), dtype=torch.int64)\n                        target['masks'] = torch.as_tensor(np.array(masks), dtype=torch.uint8)\n                        target['image_id'] = torch.tensor([idx])\n                        target['area'] = (target['boxes'][:, 3] - target['boxes'][:, 1]) * \\\n                                       (target['boxes'][:, 2] - target['boxes'][:, 0])\n                        target['iscrowd'] = torch.zeros((len(boxes),), dtype=torch.int64)\n                        return img, target\n            except:\n                pass\n\n        # Empty target\n        target['boxes'] = torch.zeros((0, 4), dtype=torch.float32)\n        target['labels'] = torch.zeros((0,), dtype=torch.int64)\n        target['masks'] = torch.zeros((0, self.img_size, self.img_size), dtype=torch.uint8)\n        target['image_id'] = torch.tensor([idx])\n        target['area'] = torch.zeros((0,), dtype=torch.float32)\n        target['iscrowd'] = torch.zeros((0,), dtype=torch.int64)\n\n        return img, target\n\n\ndef collate_fn(batch):\n    return tuple(zip(*batch))\n\n\n# ============================================================\n# PART 3: MODEL AND TRAINING\n# ============================================================\n\ndef get_maskrcnn_model(num_classes=2):\n    weights_mode = os.environ.get(\"MASKRCNN_PRETRAINED\", \"none\").lower()\n    weights_arg = None\n    if weights_mode == \"default\":\n        from torchvision.models.detection import MaskRCNN_ResNet50_FPN_V2_Weights\n        weights_arg = MaskRCNN_ResNet50_FPN_V2_Weights.DEFAULT\n    model = maskrcnn_resnet50_fpn_v2(weights=weights_arg)\n    in_features = model.roi_heads.box_predictor.cls_score.in_features\n    model.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n    in_features_mask = model.roi_heads.mask_predictor.conv5_mask.in_channels\n    model.roi_heads.mask_predictor = MaskRCNNPredictor(in_features_mask, 256, num_classes)\n    checkpoint_path = os.environ.get(\"MASKRCNN_WEIGHTS_PATH\")\n    if checkpoint_path and Path(checkpoint_path).exists():\n        print(f\"Loading Mask R-CNN checkpoint from {checkpoint_path}\")\n        state = torch.load(checkpoint_path, map_location=\"cpu\")\n        state_dict = state.get(\"model\", state.get(\"model_state_dict\", state))\n        try:\n            model.load_state_dict(state_dict, strict=False)\n        except RuntimeError as err:\n            print(f\"Warning: partial load of checkpoint failed ({err}). Continuing with available weights.\")\n    return model\n\n\ndef train_epoch(model, data_loader, optimizer, device):\n    model.train()\n    total_loss = 0\n    for images, targets in tqdm(data_loader, desc=\"Training\"):\n        images = [img.to(device) for img in images]\n        targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n        loss_dict = model(images, targets)\n        losses = sum(loss for loss in loss_dict.values())\n        optimizer.zero_grad()\n        losses.backward()\n        optimizer.step()\n        total_loss += losses.item()\n    return total_loss / len(data_loader)\n\n\ndef rle_encode(mask):\n    pixels = mask.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return runs.tolist()\n\n\ndef predict_with_tta(model, img_tensor, device, original_shape, img_size=256, threshold=0.5):\n    \"\"\"\n    Test-time augmentation with 4 transforms\n    Returns combined mask or None if authentic\n    \"\"\"\n    model.eval()\n    all_masks = []\n\n    with torch.no_grad():\n        # 1. Original\n        out = model([img_tensor.to(device)])[0]\n        if len(out['masks']) > 0 and out['scores'].max() > threshold:\n            mask = out['masks'][out['scores'] > threshold]\n            combined = torch.zeros((img_size, img_size), dtype=torch.float32, device=device)\n            for m in mask:\n                combined = torch.maximum(combined, m[0])\n            all_masks.append(combined)\n\n        # 2. Horizontal flip\n        img_h = torch.flip(img_tensor, [2])\n        out_h = model([img_h.to(device)])[0]\n        if len(out_h['masks']) > 0 and out_h['scores'].max() > threshold:\n            mask_h = out_h['masks'][out_h['scores'] > threshold]\n            combined_h = torch.zeros((img_size, img_size), dtype=torch.float32, device=device)\n            for m in mask_h:\n                combined_h = torch.maximum(combined_h, m[0])\n            combined_h = torch.flip(combined_h, [1])\n            all_masks.append(combined_h)\n\n        # 3. Vertical flip\n        img_v = torch.flip(img_tensor, [1])\n        out_v = model([img_v.to(device)])[0]\n        if len(out_v['masks']) > 0 and out_v['scores'].max() > threshold:\n            mask_v = out_v['masks'][out_v['scores'] > threshold]\n            combined_v = torch.zeros((img_size, img_size), dtype=torch.float32, device=device)\n            for m in mask_v:\n                combined_v = torch.maximum(combined_v, m[0])\n            combined_v = torch.flip(combined_v, [0])\n            all_masks.append(combined_v)\n\n        # 4. Both flips\n        img_hv = torch.flip(img_tensor, [1, 2])\n        out_hv = model([img_hv.to(device)])[0]\n        if len(out_hv['masks']) > 0 and out_hv['scores'].max() > threshold:\n            mask_hv = out_hv['masks'][out_hv['scores'] > threshold]\n            combined_hv = torch.zeros((img_size, img_size), dtype=torch.float32, device=device)\n            for m in mask_hv:\n                combined_hv = torch.maximum(combined_hv, m[0])\n            combined_hv = torch.flip(combined_hv, [0, 1])\n            all_masks.append(combined_hv)\n\n    if len(all_masks) == 0:\n        return None\n\n    avg_mask = torch.stack(all_masks).mean(dim=0)\n    avg_mask_resized = cv2.resize(\n        avg_mask.cpu().numpy(),\n        (original_shape[1], original_shape[0])\n    )\n    return (avg_mask_resized > 0.5).astype(np.uint8)\n\n\ndef predict_test_set(model, test_path, device, img_size=256, threshold=0.5, use_tta=True):\n    model.eval()\n    predictions = {}\n    test_files = sorted([f for f in os.listdir(test_path)\n                        if f.lower().endswith(('.png', '.jpg', '.jpeg'))])\n\n    with torch.no_grad():\n        for file in tqdm(test_files, desc=\"Predicting\"):\n            case_id = int(file.split('.')[0])\n            img = cv2.imread(os.path.join(test_path, file))\n            original_shape = img.shape[:2]\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            img_resized = cv2.resize(img, (img_size, img_size))\n            img_tensor = torch.from_numpy(img_resized.astype(np.float32) / 255.0).permute(2, 0, 1)\n\n            if use_tta:\n                final_mask = predict_with_tta(model, img_tensor, device, original_shape, img_size, threshold)\n            else:\n                outputs = model([img_tensor.to(device)])[0]\n                if len(outputs['masks']) > 0 and outputs['scores'].max() > threshold:\n                    high_score_idx = outputs['scores'] > threshold\n                    if high_score_idx.sum() > 0:\n                        masks = outputs['masks'][high_score_idx]\n                        combined_mask = torch.zeros((img_size, img_size), dtype=torch.float32).to(device)\n                        for mask in masks:\n                            combined_mask = torch.maximum(combined_mask, mask[0])\n                        combined_mask = cv2.resize(combined_mask.cpu().numpy(),\n                                                  (original_shape[1], original_shape[0]))\n                        final_mask = (combined_mask > 0.5).astype(np.uint8)\n                    else:\n                        final_mask = None\n                else:\n                    final_mask = None\n\n            if final_mask is not None and final_mask.sum() > 100:\n                kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (3, 3))\n                final_mask = cv2.morphologyEx(final_mask, cv2.MORPH_CLOSE, kernel)\n                final_mask = cv2.morphologyEx(final_mask, cv2.MORPH_OPEN, kernel)\n                predictions[case_id] = json.dumps(rle_encode(final_mask))\n            else:\n                predictions[case_id] = \"authentic\"\n\n    return predictions\n\n\n# ============================================================\n# MAIN PIPELINE\n# ============================================================\n\ndef main():\n    print(\"=\"*60)\n    print(\"MASK R-CNN WITH SYNTHETIC DATA - KAGGLE VERSION\")\n    print(\"=\"*60)\n\n    # Configuration\n    IMG_SIZE = int(os.environ.get(\"MASKRCNN_IMG_SIZE\", 256))\n    BATCH_SIZE = int(os.environ.get(\"MASKRCNN_BATCH_SIZE\", 8))  # Will adjust based on device\n    NUM_EPOCHS = int(os.environ.get(\"MASKRCNN_EPOCHS\", 5))\n    LEARNING_RATE = float(os.environ.get(\"MASKRCNN_LR\", 0.001))\n    NUM_SYNTHETIC = int(os.environ.get(\"MASKRCNN_SYNTHETIC\", 1000))  # Reduced from 2000 for Kaggle memory safety\n    THRESHOLD = float(os.environ.get(\"MASKRCNN_THRESHOLD\", 0.5))\n\n    use_kaggle_layout = os.path.exists('/kaggle/input')\n    if use_kaggle_layout:\n        base_path = '/kaggle/input/recodai-luc-scientific-image-forgery-detection'\n        synthetic_base = '/kaggle/working/synthetic'\n    else:\n        raw_candidate = Path('./data/raw/train_images')\n        default_candidate = Path('./data/train_images')\n        alt_bundle = Path('./recodai-luc-scientific-image-forgery-detection/train_images')\n        if raw_candidate.exists():\n            base_path = './data/raw'\n        elif default_candidate.exists():\n            base_path = './data'\n        elif alt_bundle.exists():\n            base_path = './recodai-luc-scientific-image-forgery-detection'\n        else:\n            base_path = './data/raw'\n        processed_candidate = Path('./data/processed/synthetic_forged')\n        alt_processed = Path('./synthetic/forged')\n        if processed_candidate.exists():\n            synthetic_base = './data/processed'\n        elif alt_processed.exists():\n            synthetic_base = './synthetic'\n        else:\n            synthetic_base = './data/synthetic'\n        BATCH_SIZE = 2  # CPU-friendly default\n\n    paths = {\n        'train_authentic': f'{base_path}/train_images/authentic',\n        'train_forged': f'{base_path}/train_images/forged',\n        'train_masks': f'{base_path}/train_masks',\n        'synthetic_forged': f'{synthetic_base}/synthetic_forged' if not use_kaggle_layout else f'{synthetic_base}/forged',\n        'synthetic_masks': f'{synthetic_base}/synthetic_masks' if not use_kaggle_layout else f'{synthetic_base}/masks',\n        'test_images': f'{base_path}/test_images'\n    }\n\n    # STEP 1: Generate synthetic data\n    print(\"\\nSTEP 1: Generating synthetic forgeries...\")\n    if use_kaggle_layout or not os.path.exists(paths['synthetic_forged']):\n        os.makedirs(paths['synthetic_forged'], exist_ok=True)\n        os.makedirs(paths['synthetic_masks'], exist_ok=True)\n        generate_synthetic_forgeries(\n            paths['train_authentic'],\n            paths['train_forged'],\n            paths['train_masks'],\n            paths['synthetic_forged'],\n            paths['synthetic_masks'],\n            num_synthetic=NUM_SYNTHETIC\n        )\n    else:\n        print(\"Synthetic directories already populated; skipping regeneration.\")\n\n    # STEP 2: Load dataset\n    print(\"\\nSTEP 2: Loading combined dataset...\")\n    train_dataset = ForgeryDatasetCombined(\n        paths['train_authentic'],\n        paths['train_forged'],\n        paths['train_masks'],\n        paths['synthetic_forged'],\n        paths['synthetic_masks'],\n        img_size=IMG_SIZE,\n        max_original=800 if use_kaggle_layout else None,\n        max_synthetic=NUM_SYNTHETIC if use_kaggle_layout else None\n    )\n\n    # Device detection: TPU > GPU > CPU\n    device = None\n\n    # Try TPU first (uses torch_xla if available)\n    try:\n        import torch_xla.core.xla_model as xm\n        device = xm.xla_device()\n        print(f\"Device: TPU (torch_xla)\")\n    except ImportError:\n        # TPU not available, try GPU\n        if torch.cuda.is_available():\n            device = torch.device('cuda')\n            print(f\"Device: cuda\")\n        else:\n            device = torch.device('cpu')\n            print(f\"Device: cpu\")\n\n    if device is None:\n        device = torch.device('cpu')\n    print(f\"Device: {device}\")\n\n    train_loader = DataLoader(\n        train_dataset,\n        batch_size=BATCH_SIZE,\n        shuffle=True,\n        num_workers=2,\n        collate_fn=collate_fn,\n        pin_memory=True\n    )\n\n    # STEP 3: Create and train model\n    print(\"\\nSTEP 3: Training Mask R-CNN...\")\n    model = get_maskrcnn_model(num_classes=2).to(device)\n    optimizer = torch.optim.Adam([p for p in model.parameters() if p.requires_grad], lr=LEARNING_RATE)\n\n    for epoch in range(NUM_EPOCHS):\n        loss = train_epoch(model, train_loader, optimizer, device)\n        print(f\"Epoch {epoch+1}/{NUM_EPOCHS} - Loss: {loss:.4f}\")\n\n    # STEP 4: Save model\n    print(\"\\nSTEP 4: Saving model...\")\n    torch.save(model.state_dict(), 'maskrcnn_synthetic.pth')\n\n    # STEP 5: Generate predictions\n    print(f\"\\nSTEP 5: Generating predictions (threshold={THRESHOLD})...\")\n    predictions = predict_test_set(model, paths['test_images'], device, IMG_SIZE, THRESHOLD)\n\n    # STEP 6: Create submission\n    print(\"\\nSTEP 6: Creating submission...\")\n    sample = pd.read_csv(f'{base_path}/sample_submission.csv')\n    submission_data = []\n    for _, row in sample.iterrows():\n        case_id = row['case_id']\n        annotation = predictions.get(case_id, \"authentic\")\n        submission_data.append({'case_id': case_id, 'annotation': annotation})\n\n    submission_df = pd.DataFrame(submission_data)\n    submission_df.to_csv('submission.csv', index=False)\n\n    print(f\"\\nSubmission saved!\")\n    print(f\"Total cases: {len(submission_df)}\")\n    print(f\"Predicted forged: {sum(1 for x in predictions.values() if x != 'authentic')}\")\n    print(\"\\n\" + \"=\"*60)\n    print(\"COMPLETE - Ready for submission!\")\n    print(\"=\"*60)\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-07T05:29:14.071042Z","iopub.execute_input":"2025-11-07T05:29:14.071385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The end ","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}