{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport torch\nimport numpy as np\nimport pandas as pd\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport warnings\nwarnings.filterwarnings('ignore')\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\nclass EnhancedUNet(nn.Module):\n    \"\"\"Enhanced U-Net with residual connections and attention\"\"\"\n    \n    def __init__(self, in_channels=3, out_channels=1):\n        super().__init__()\n        \n        # Encoder with residual blocks\n        self.enc1 = ResidualBlock(in_channels, 32)\n        self.enc2 = ResidualBlock(32, 64)\n        self.enc3 = ResidualBlock(64, 128)\n        self.enc4 = ResidualBlock(128, 256)\n        \n        # Bottleneck with attention\n        self.bottleneck = ResidualBlock(256, 512)\n        self.attention = SpatialAttention(512)\n        \n        # Decoder with skip connections and attention\n        self.up4 = UpsampleBlock(512, 256)\n        self.dec4 = ResidualBlock(512, 256)\n        \n        self.up3 = UpsampleBlock(256, 128)\n        self.dec3 = ResidualBlock(256, 128)\n        \n        self.up2 = UpsampleBlock(128, 64)\n        self.dec2 = ResidualBlock(128, 64)\n        \n        self.up1 = UpsampleBlock(64, 32)\n        self.dec1 = ResidualBlock(64, 32)\n        \n        # Output with multi-scale features\n        self.out_conv = nn.Conv2d(32, out_channels, 1)\n        \n        # Deep supervision\n        self.ds3 = nn.Conv2d(128, out_channels, 1)\n        self.ds2 = nn.Conv2d(64, out_channels, 1)\n        \n        self.pool = nn.MaxPool2d(2, 2)\n    \n    def forward(self, x):\n        # Encoder\n        e1 = self.enc1(x)\n        e2 = self.enc2(self.pool(e1))\n        e3 = self.enc3(self.pool(e2))\n        e4 = self.enc4(self.pool(e3))\n        \n        # Bottleneck with attention\n        b = self.bottleneck(self.pool(e4))\n        b = self.attention(b)\n        \n        # Decoder with deep supervision\n        d4 = self.up4(b)\n        d4 = torch.cat([d4, e4], dim=1)\n        d4 = self.dec4(d4)\n        \n        d3 = self.up3(d4)\n        d3 = torch.cat([d3, e3], dim=1)\n        d3 = self.dec3(d3)\n        ds3 = torch.sigmoid(self.ds3(d3))\n        \n        d2 = self.up2(d3)\n        d2 = torch.cat([d2, e2], dim=1)\n        d2 = self.dec2(d2)\n        ds2 = torch.sigmoid(self.ds2(d2))\n        \n        d1 = self.up1(d2)\n        d1 = torch.cat([d1, e1], dim=1)\n        d1 = self.dec1(d1)\n        \n        out = torch.sigmoid(self.out_conv(d1))\n        \n        # Multi-scale output for better training\n        return out, ds2, ds3\n\nclass ResidualBlock(nn.Module):\n    def __init__(self, in_ch, out_ch):\n        super().__init__()\n        self.conv1 = nn.Conv2d(in_ch, out_ch, 3, padding=1)\n        self.bn1 = nn.BatchNorm2d(out_ch)\n        self.conv2 = nn.Conv2d(out_ch, out_ch, 3, padding=1)\n        self.bn2 = nn.BatchNorm2d(out_ch)\n        \n        if in_ch != out_ch:\n            self.shortcut = nn.Conv2d(in_ch, out_ch, 1)\n        else:\n            self.shortcut = nn.Identity()\n    \n    def forward(self, x):\n        residual = self.shortcut(x)\n        out = F.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        out += residual\n        return F.relu(out)\n\nclass SpatialAttention(nn.Module):\n    def __init__(self, channels):\n        super().__init__()\n        self.conv = nn.Conv2d(channels, 1, 1)\n        self.sigmoid = nn.Sigmoid()\n    \n    def forward(self, x):\n        attn = self.sigmoid(self.conv(x))\n        return x * attn + x\n\nclass UpsampleBlock(nn.Module):\n    def __init__(self, in_ch, out_ch):\n        super().__init__()\n        self.up = nn.ConvTranspose2d(in_ch, out_ch, 2, 2)\n    \n    def forward(self, x):\n        return self.up(x)\n\nclass DiceBCELoss(nn.Module):\n    def __init__(self, weight=None, size_average=True):\n        super().__init__()\n    \n    def forward(self, inputs, targets, smooth=1):\n        # Flatten label and prediction tensors\n        inputs = inputs.view(-1)\n        targets = targets.view(-1)\n        \n        intersection = (inputs * targets).sum()\n        dice_loss = 1 - (2. * intersection + smooth) / (inputs.sum() + targets.sum() + smooth)\n        BCE = F.binary_cross_entropy(inputs, targets, reduction='mean')\n        \n        return BCE + dice_loss\n\nclass EnhancedDataset(Dataset):\n    def __init__(self, authentic_path, forged_path, masks_path, \n                 img_size=256, is_train=True, augment=True):\n        self.img_size = img_size\n        self.is_train = is_train\n        self.augment = augment and is_train\n        self.samples = []\n        \n        # Enhanced data collection with balance\n        authentic_files = []\n        forged_files = []\n        \n        if os.path.exists(authentic_path):\n            authentic_files = [f for f in os.listdir(authentic_path) \n                             if f.lower().endswith(('.png', '.jpg', '.jpeg'))]\n        \n        if os.path.exists(forged_path):\n            forged_files = [f for f in os.listdir(forged_path) \n                          if f.lower().endswith(('.png', '.jpg', '.jpeg'))]\n        \n        # Balance the dataset\n        max_samples = min(len(authentic_files), len(forged_files), 800 if is_train else 100)\n        \n        for file in authentic_files[:max_samples]:\n            img_path = os.path.join(authentic_path, file)\n            mask_path = os.path.join(masks_path, f\"{file.split('.')[0]}.npy\")\n            self.samples.append((img_path, mask_path, 0))\n        \n        for file in forged_files[:max_samples]:\n            img_path = os.path.join(forged_path, file)\n            mask_path = os.path.join(masks_path, f\"{file.split('.')[0]}.npy\")\n            self.samples.append((img_path, mask_path, 1))\n        \n        # Data augmentation\n        if self.augment:\n            self.transform = A.Compose([\n                A.HorizontalFlip(p=0.5),\n                A.VerticalFlip(p=0.3),\n                A.RandomRotate90(p=0.3),\n                A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=15, p=0.3),\n                A.RandomBrightnessContrast(p=0.2),\n                A.GaussNoise(p=0.2),\n                A.Blur(blur_limit=3, p=0.2),\n                A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                ToTensorV2(),\n            ])\n        else:\n            self.transform = A.Compose([\n                A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                ToTensorV2(),\n            ])\n        \n        self.mask_transform = A.Compose([\n            A.Resize(img_size, img_size, interpolation=cv2.INTER_NEAREST),\n        ])\n        \n        print(f\"Loaded {len(self.samples)} samples (balanced)\")\n    \n    def __len__(self):\n        return len(self.samples)\n    \n    def __getitem__(self, idx):\n        img_path, mask_path, is_forged = self.samples[idx]\n        \n        # Load image\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Load and process mask\n        if is_forged and os.path.exists(mask_path):\n            try:\n                mask = np.load(mask_path)\n                if mask.ndim == 3:\n                    mask = mask.max(axis=0) if mask.shape[0] <= 10 else mask.max(axis=-1)\n                mask = (mask > 0).astype(np.float32)\n            except:\n                mask = np.zeros(img.shape[:2], dtype=np.float32)\n        else:\n            mask = np.zeros(img.shape[:2], dtype=np.float32)\n        \n        # Apply augmentations\n        if self.augment:\n            transformed = self.transform(image=img, mask=mask)\n            img = transformed['image']\n            mask = transformed['mask']\n        else:\n            # Resize without augmentation\n            img = cv2.resize(img, (self.img_size, self.img_size))\n            img = img.astype(np.float32) / 255.0\n            img = torch.from_numpy(img).permute(2, 0, 1).float()\n            \n            mask = cv2.resize(mask, (self.img_size, self.img_size), interpolation=cv2.INTER_NEAREST)\n            mask = torch.from_numpy(mask).unsqueeze(0).float()\n        \n        return img, mask\n\ndef enhanced_rle_encode(mask):\n    \"\"\"Enhanced RLE encoding with better compression\"\"\"\n    if not isinstance(mask, np.ndarray):\n        mask = np.array(mask)\n    \n    mask = (mask > 0.5).astype(np.uint8)\n    \n    if mask.sum() == 0:\n        return \"authentic\"\n    \n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    \n    return ' '.join(str(x) for x in runs)\n\ndef calculate_iou(pred, target):\n    \"\"\"Calculate Intersection over Union\"\"\"\n    pred = (pred > 0.5).float()\n    target = (target > 0.5).float()\n    \n    intersection = (pred * target).sum()\n    union = pred.sum() + target.sum() - intersection\n    \n    return intersection / (union + 1e-6)\n\ndef train_enhanced(model, train_loader, val_loader, optimizer, criterion, scheduler, device):\n    model.train()\n    train_loss = 0\n    train_iou = 0\n    \n    for imgs, masks in tqdm(train_loader, desc=\"Training\", leave=False):\n        imgs, masks = imgs.to(device), masks.to(device)\n        \n        optimizer.zero_grad()\n        outputs, ds2, ds3 = model(imgs)\n        \n        # Multi-scale loss\n        loss_main = criterion(outputs, masks)\n        loss_ds2 = criterion(ds2, F.interpolate(masks, scale_factor=0.5, mode='bilinear'))\n        loss_ds3 = criterion(ds3, F.interpolate(masks, scale_factor=0.25, mode='bilinear'))\n        \n        loss = loss_main + 0.5 * loss_ds2 + 0.3 * loss_ds3\n        \n        loss.backward()\n        torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n        optimizer.step()\n        \n        train_loss += loss_main.item()\n        train_iou += calculate_iou(outputs, masks).item()\n    \n    # Validation\n    model.eval()\n    val_loss = 0\n    val_iou = 0\n    \n    with torch.no_grad():\n        for imgs, masks in val_loader:\n            imgs, masks = imgs.to(device), masks.to(device)\n            outputs, _, _ = model(imgs)\n            loss = criterion(outputs, masks)\n            val_loss += loss.item()\n            val_iou += calculate_iou(outputs, masks).item()\n    \n    train_loss /= len(train_loader)\n    train_iou /= len(train_loader)\n    val_loss /= len(val_loader)\n    val_iou /= len(val_loader)\n    \n    scheduler.step(val_loss)\n    \n    return train_loss, train_iou, val_loss, val_iou\n\ndef predict_enhanced(model, test_path, device, img_size=256):\n    model.eval()\n    predictions = {}\n    \n    test_files = [f for f in os.listdir(test_path) \n                  if f.lower().endswith(('.png', '.jpg', '.jpeg'))]\n    \n    # Test time augmentation\n    tta_transforms = [\n        A.HorizontalFlip(p=1.0),\n        A.VerticalFlip(p=1.0),\n        A.Rotate(limit=90, p=1.0),\n    ]\n    \n    base_transform = A.Compose([\n        A.Resize(img_size, img_size),\n        A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ToTensorV2(),\n    ])\n    \n    with torch.no_grad():\n        for file in tqdm(test_files, desc=\"Predicting\"):\n            case_id = file.split('.')[0]\n            img_path = os.path.join(test_path, file)\n            \n            # Load original image\n            img = cv2.imread(img_path)\n            original_size = img.shape[:2]\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            \n            # Base prediction\n            img_resized = cv2.resize(img, (img_size, img_size))\n            img_tensor = base_transform(image=img_resized)['image']\n            img_tensor = img_tensor.unsqueeze(0).to(device)\n            \n            mask_pred, _, _ = model(img_tensor)\n            mask_pred = mask_pred[0, 0].cpu().numpy()\n            \n            # TTA predictions\n            tta_preds = [mask_pred]\n            \n            for transform in tta_transforms:\n                augmented = transform(image=img_resized)['image']\n                augmented_tensor = base_transform(image=augmented)['image']\n                augmented_tensor = augmented_tensor.unsqueeze(0).to(device)\n                \n                pred, _, _ = model(augmented_tensor)\n                pred = pred[0, 0].cpu().numpy()\n                \n                # Reverse augmentation\n                if transform == tta_transforms[0]:  # Horizontal flip\n                    pred = cv2.flip(pred, 1)\n                elif transform == tta_transforms[1]:  # Vertical flip\n                    pred = cv2.flip(pred, 0)\n                elif transform == tta_transforms[2]:  # Rotate\n                    pred = cv2.rotate(pred, cv2.ROTATE_90_COUNTERCLOCKWISE)\n                \n                tta_preds.append(pred)\n            \n            # Ensemble TTA predictions\n            mask_pred = np.mean(tta_preds, axis=0)\n            \n            # Post-processing\n            mask_pred = (mask_pred > 0.3).astype(np.uint8)\n            mask_pred = cv2.resize(mask_pred, (original_size[1], original_size[0]), \n                                  interpolation=cv2.INTER_NEAREST)\n            \n            # Enhanced morphological operations\n            kernel = np.ones((5, 5), np.uint8)\n            mask_pred = cv2.morphologyEx(mask_pred, cv2.MORPH_OPEN, kernel)\n            mask_pred = cv2.morphologyEx(mask_pred, cv2.MORPH_CLOSE, kernel)\n            \n            # Remove small regions\n            contours, _ = cv2.findContours(mask_pred, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n            for contour in contours:\n                if cv2.contourArea(contour) < 500:\n                    cv2.drawContours(mask_pred, [contour], 0, 0, -1)\n            \n            # Encode\n            if mask_pred.sum() < 300:  # Adjusted threshold\n                predictions[case_id] = \"authentic\"\n            else:\n                predictions[case_id] = enhanced_rle_encode(mask_pred)\n    \n    return predictions\n\ndef main():\n    print(\"=\"*60)\n    print(\"ENHANCED FORGERY DETECTION WITH HIGH ACCURACY\")\n    print(\"=\"*60)\n    \n    # Paths\n    base_path = '/kaggle/input/recodai-luc-scientific-image-forgery-detection'\n    paths = {\n        'train_authentic': f'{base_path}/train_images/authentic',\n        'train_forged': f'{base_path}/train_images/forged',\n        'train_masks': f'{base_path}/train_masks',\n        'test_images': f'{base_path}/test_images'\n    }\n    \n    # Enhanced hyperparameters\n    IMG_SIZE = 256  # Increased for better details\n    BATCH_SIZE = 8 if device.type == 'cuda' else 4\n    NUM_EPOCHS = 10  # More epochs with early stopping\n    LR = 0.001\n    PATIENCE = 3\n    \n    print(f\"\\nConfig: {IMG_SIZE}x{IMG_SIZE}, BS={BATCH_SIZE}, Epochs={NUM_EPOCHS}\")\n    \n    # Create datasets with train/val split\n    print(\"\\n[1/6] Loading and splitting data...\")\n    full_dataset = EnhancedDataset(\n        paths['train_authentic'],\n        paths['train_forged'],\n        paths['train_masks'],\n        img_size=IMG_SIZE,\n        is_train=True,\n        augment=True\n    )\n    \n    # Split dataset\n    train_size = int(0.8 * len(full_dataset))\n    val_size = len(full_dataset) - train_size\n    train_dataset, val_dataset = torch.utils.data.random_split(\n        full_dataset, [train_size, val_size]\n    )\n    \n    # Use augmentation only for training\n    train_dataset.dataset.augment = True\n    val_dataset.dataset.augment = False\n    \n    train_loader = DataLoader(\n        train_dataset,\n        batch_size=BATCH_SIZE,\n        shuffle=True,\n        num_workers=2,\n        pin_memory=True\n    )\n    \n    val_loader = DataLoader(\n        val_dataset,\n        batch_size=BATCH_SIZE,\n        shuffle=False,\n        num_workers=2,\n        pin_memory=True\n    )\n    \n    # Enhanced model\n    print(\"\\n[2/6] Creating enhanced model...\")\n    model = EnhancedUNet(in_channels=3, out_channels=1).to(device)\n    \n    params = sum(p.numel() for p in model.parameters())\n    print(f\"Model parameters: {params:,}\")\n    \n    # Enhanced training setup\n    criterion = DiceBCELoss()\n    optimizer = torch.optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)\n    scheduler = ReduceLROnPlateau(optimizer, mode='min', patience=2, factor=0.5, verbose=True)\n    \n    # Training with early stopping\n    print(f\"\\n[3/6] Training for {NUM_EPOCHS} epochs...\")\n    best_val_loss = float('inf')\n    patience_counter = 0\n    \n    for epoch in range(NUM_EPOCHS):\n        train_loss, train_iou, val_loss, val_iou = train_enhanced(\n            model, train_loader, val_loader, optimizer, criterion, scheduler, device\n        )\n        \n        print(f\"Epoch {epoch+1}/{NUM_EPOCHS}\")\n        print(f\"  Train Loss: {train_loss:.4f}, Train IoU: {train_iou:.4f}\")\n        print(f\"  Val Loss: {val_loss:.4f}, Val IoU: {val_iou:.4f}\")\n        \n        # Early stopping\n        if val_loss < best_val_loss:\n            best_val_loss = val_loss\n            patience_counter = 0\n            torch.save(model.state_dict(), 'best_model.pth')\n            print(\"  ↳ Saved best model!\")\n        else:\n            patience_counter += 1\n            if patience_counter >= PATIENCE:\n                print(\"  ↳ Early stopping!\")\n                break\n    \n    # Load best model\n    model.load_state_dict(torch.load('best_model.pth'))\n    \n    # Save final model\n    print(\"\\n[4/6] Saving final model...\")\n    torch.save(model.state_dict(), 'enhanced_model.pth')\n    \n    # Enhanced prediction\n    print(\"\\n[5/6] Predicting on test set...\")\n    predictions = predict_enhanced(model, paths['test_images'], device, IMG_SIZE)\n    \n    # Create submission\n    print(\"\\n[6/6] Creating submission file...\")\n    sample = pd.read_csv(f'{base_path}/sample_submission.csv')\n    submission_data = []\n    \n    for case_id in sample['case_id']:\n        annotation = predictions.get(str(case_id), \"authentic\")\n        submission_data.append({'case_id': case_id, 'annotation': annotation})\n    \n    submission = pd.DataFrame(submission_data)\n    submission.to_csv('submission.csv', index=False)\n    \n    # Enhanced stats\n    authentic = (submission['annotation'] == 'authentic').sum()\n    forged = len(submission) - authentic\n    forged_with_mask = sum(1 for x in submission['annotation'] if x != 'authentic')\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"ENHANCED MODEL COMPLETE! ✓\")\n    print(\"=\"*60)\n    print(f\"Predictions: {len(submission)}\")\n    print(f\"  Authentic: {authentic}\")\n    print(f\"  Forged: {forged}\")\n    print(f\"  Forged with masks: {forged_with_mask}\")\n    print(f\"Submission saved: enhanced_submission.csv\")\n    print(\"=\"*60)\n\nif __name__ == '__main__':\n    import time\n    start = time.time()\n    main()\n    elapsed = time.time() - start\n    print(f\"\\nTotal time: {elapsed:.1f}s ({elapsed/60:.1f} min)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T19:00:25.324935Z","iopub.execute_input":"2025-10-27T19:00:25.325572Z","iopub.status.idle":"2025-10-27T19:09:45.812843Z","shell.execute_reply.started":"2025-10-27T19:00:25.325545Z","shell.execute_reply":"2025-10-27T19:09:45.811776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}