{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"sourceType":"competition"},{"sourceId":14167480,"sourceType":"datasetVersion","datasetId":9030673},{"sourceId":4534,"sourceType":"modelInstanceVersion","modelInstanceId":3326,"modelId":986},{"sourceId":686586,"sourceType":"modelInstanceVersion","modelInstanceId":520737,"modelId":534998}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport math\nimport torch\nimport numpy as np\nimport pandas as pd\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom PIL import Image\nfrom pathlib import Path\nfrom transformers import AutoImageProcessor, AutoModel\n\nclass CONFIG:\n    test_images_path = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images\"\n    sample_sub_path = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv\"\n    model1_path = \"/kaggle/input/modelsbest309base/best_model.pth\"\n    model2_path = \"/kaggle/input/dinobestmodel/pytorch/default/1/dino197.pth\"\n    dino_path = \"/kaggle/input/dinov2/pytorch/base/1\"\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    img_size = 512\n    use_tta = True\n    min_area_percent = 0.05 # percent new!\n    min_confidence = 0.336  # better conf\n\nclass Decoder(nn.Module):\n    \n    def __init__(self, in_ch=768, out_ch=1):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Conv2d(in_ch, 256, 3, padding=1), nn.ReLU(),\n            nn.Conv2d(256, 64, 3, padding=1), nn.ReLU(),\n            nn.Conv2d(64, out_ch, 1)\n        )\n    \n    def forward(self, f, size):\n        return self.net(F.interpolate(f, size=size, mode=\"bilinear\", align_corners=False))\n\nclass DinoSegmenter(nn.Module):\n    \n    def __init__(self, encoder, processor):\n        super().__init__()\n        self.encoder = encoder\n        self.processor = processor\n        self.seg_head = Decoder(768, 1)\n    \n    def forward_features(self, x):\n        imgs = (x*255).clamp(0, 255).byte().permute(0, 2, 3, 1).cpu().numpy()\n        inputs = self.processor(images=list(imgs), return_tensors=\"pt\").to(x.device)\n        \n        with torch.no_grad():\n            feats = self.encoder(**inputs).last_hidden_state\n        \n        B, N, C = feats.shape\n        fmap = feats[:, 1:, :].permute(0, 2, 1)\n        \n        s = int(math.sqrt(N-1))\n        fmap = fmap.reshape(B, C, s, s)\n        \n        return fmap\n    \n    def forward_seg(self, x):\n        fmap = self.forward_features(x)\n        return self.seg_head(fmap, (CONFIG.img_size, CONFIG.img_size))\n\nclass Model(nn.Module):\n    \n    def __init__(self):\n        super().__init__()\n        self.encoder = nn.Sequential(\n            nn.Conv2d(3, 32, 3, padding=1), nn.ReLU(),\n            nn.Conv2d(32, 32, 3, padding=1), nn.ReLU(),\n            nn.MaxPool2d(2),\n            nn.Conv2d(32, 64, 3, padding=1), nn.ReLU(),\n            nn.Conv2d(64, 64, 3, padding=1), nn.ReLU(),\n            nn.MaxPool2d(2),\n        )\n        \n        self.decoder = nn.Sequential(\n            nn.Conv2d(64, 32, 3, padding=1), nn.ReLU(),\n            nn.Upsample(scale_factor=2, mode='bilinear'),\n            nn.Conv2d(32, 32, 3, padding=1), nn.ReLU(),\n            nn.Upsample(scale_factor=2, mode='bilinear'),\n            nn.Conv2d(32, 1, 1),\n        )\n    \n    def forward(self, x):\n        x = self.encoder(x)\n        x = self.decoder(x)\n        x = F.interpolate(x, size=(CONFIG.img_size, CONFIG.img_size), mode='bilinear', align_corners=False)\n        \n        return x\n\ndef load_model(model_path):\n    try:\n        checkpoint = torch.load(model_path, map_location=CONFIG.device)\n        \n        if isinstance(checkpoint, dict):\n            state_dict = None\n            if 'model_state_dict' in checkpoint:\n                state_dict = checkpoint['model_state_dict']\n            elif 'state_dict' in checkpoint:\n                state_dict = checkpoint['state_dict']\n            elif 'model' in checkpoint:\n                model = checkpoint['model']\n                if hasattr(model, 'eval'):\n                    model.eval()\n                return model.to(CONFIG.device)\n            else:\n                state_dict = checkpoint\n            \n            try:\n                processor = AutoImageProcessor.from_pretrained(CONFIG.dino_path, local_files_only=True)\n                encoder = AutoModel.from_pretrained(CONFIG.dino_path, local_files_only=True).eval().to(CONFIG.device)\n                model = DinoSegmenter(encoder, processor).to(CONFIG.device)\n                \n                if state_dict is not None:\n                    model.load_state_dict(state_dict, strict=False)\n                \n                model.eval()\n                return model\n            except:\n                try:\n                    model = Model().to(CONFIG.device)\n                    if state_dict is not None:\n                        model.load_state_dict(state_dict, strict=False)\n                    model.eval()\n                    return model\n                except:\n                    return None\n        \n        elif hasattr(checkpoint, 'eval'):\n            checkpoint.eval()\n            return checkpoint.to(CONFIG.device)\n        \n        return None\n    except Exception as e:\n        print(f\"Error {Path(model_path).name}: {e}\")\n        return None\n\ndef predict_with_tta(model, image_tensor):\n    predictions = []\n    \n    with torch.no_grad():\n        if hasattr(model, 'forward_seg'):\n            pred = torch.sigmoid(model.forward_seg(image_tensor))\n        else:\n            pred = torch.sigmoid(model(image_tensor))\n    \n    predictions.append(pred)\n    \n    with torch.no_grad():\n        if hasattr(model, 'forward_seg'):\n            pred = torch.sigmoid(model.forward_seg(torch.flip(image_tensor, dims=[3])))\n        else:\n            pred = torch.sigmoid(model(torch.flip(image_tensor, dims=[3])))\n    \n    predictions.append(torch.flip(pred, dims=[3]))\n    \n    with torch.no_grad():\n        if hasattr(model, 'forward_seg'):\n            pred = torch.sigmoid(model.forward_seg(torch.flip(image_tensor, dims=[2])))\n        else:\n            pred = torch.sigmoid(model(torch.flip(image_tensor, dims=[2])))\n    \n    predictions.append(torch.flip(pred, dims=[2]))\n    \n    if CONFIG.use_tta:\n        with torch.no_grad():\n            if hasattr(model, 'forward_seg'):\n                pred = torch.sigmoid(model.forward_seg(torch.rot90(image_tensor, 1, [2, 3])))\n            else:\n                pred = torch.sigmoid(model(torch.rot90(image_tensor, 1, [2, 3])))\n        \n        predictions.append(torch.rot90(pred, -1, [2, 3]))\n        \n        return torch.stack(predictions).mean(0)[0, 0].detach().cpu().numpy()\n    else:\n        return predictions[0][0, 0].detach().cpu().numpy()\n\ndef postprocess(pred, original_size):\n    pred = cv2.GaussianBlur(pred, (3, 3), 0)\n    mean_val = np.mean(pred)\n    std_val = np.std(pred)\n    thr = mean_val + 0.3 * std_val\n    mask = (pred > thr).astype(np.uint8)\n    \n    if mask.sum() > 0:\n        num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(mask, connectivity=8)\n        for i in range(1, num_labels):\n            if stats[i, cv2.CC_STAT_AREA] < 30:\n                mask[labels == i] = 0\n        \n        mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5, 5), np.uint8))\n        mask = cv2.morphologyEx(mask, cv2.MORPH_OPEN, np.ones((3, 3), np.uint8))\n    \n    mask = cv2.resize(mask, original_size, interpolation=cv2.INTER_NEAREST)\n    return mask\n\ndef rle_encode(mask):\n    pixels = mask.T.flatten()\n    dots = np.where(pixels == 1)[0]\n    \n    if len(dots) == 0:\n        return \"authentic\"\n    \n    run_lengths = []\n    prev = -2\n    \n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    \n    return json.dumps([int(x) for x in run_lengths])\n\nmodel1 = load_model(CONFIG.model1_path)\nmodel2 = load_model(CONFIG.model2_path)\n\nmodels = {}\nif model1:\n    models['model1'] = model1\n    print(f\"model 1 success!\")\nif model2:\n    models['model2'] = model2\n    print(f\"model 2 success!\")\n\npredictions = []\nimage_files = sorted([f for f in os.listdir(CONFIG.test_images_path) if f.lower().endswith(('.png', '.jpg', '.jpeg', '.tiff', '.bmp'))])\n\nfor image_name in image_files:\n    image_path = Path(CONFIG.test_images_path) / image_name\n    image = Image.open(image_path).convert(\"RGB\")\n    \n    original_width, original_height = image.size\n    total_pixels = original_width * original_height\n    \n    min_pixels_threshold = int(total_pixels * CONFIG.min_area_percent / 100.0)\n    \n    image_array = np.array(image.resize((CONFIG.img_size, CONFIG.img_size)), np.float32) / 255\n    image_tensor = torch.from_numpy(image_array).permute(2, 0, 1)[None].to(CONFIG.device)\n    \n    ensemble_preds = []\n    for model in models.values():\n        pred = predict_with_tta(model, image_tensor)\n        ensemble_preds.append(pred)\n    \n    final_pred = np.mean(ensemble_preds, axis=0) if ensemble_preds else np.zeros((CONFIG.img_size, CONFIG.img_size))\n        \n    mask = postprocess(final_pred, (original_width, original_height))\n    mask_pixels = int(mask.sum())\n    \n    if mask_pixels > 0:\n        mask_resized = cv2.resize(mask, (CONFIG.img_size, CONFIG.img_size), interpolation=cv2.INTER_NEAREST)\n        mean_inside = float(final_pred[mask_resized == 1].mean()) if (mask_resized == 1).any() else 0.0\n    else:\n        mean_inside = 0.0\n        \n    if mask_pixels < min_pixels_threshold or mean_inside < CONFIG.min_confidence:\n        annotation = \"authentic\"\n    else:\n        annotation = rle_encode(mask)\n        \n    area_percent = (mask_pixels / total_pixels) * 100 if total_pixels > 0 else 0\n        \n    predictions.append({\n        \"case_id\": Path(image_name).stem,\n        \"annotation\": annotation\n    })\n\npredictions_df = pd.DataFrame(predictions)\npredictions_df[\"case_id\"] = predictions_df[\"case_id\"].astype(str)\n    \nsubmission = pd.read_csv(CONFIG.sample_sub_path)\nsubmission[\"case_id\"] = submission[\"case_id\"].astype(str)\nsubmission = submission[[\"case_id\"]].merge(predictions_df[[\"case_id\", \"annotation\"]], on=\"case_id\", how=\"left\")\n\nsubmission[\"annotation\"] = submission[\"annotation\"].fillna(\"authentic\")\nsubmission[[\"case_id\", \"annotation\"]].to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T19:09:06.660685Z","iopub.execute_input":"2025-12-27T19:09:06.661173Z","iopub.status.idle":"2025-12-27T19:09:49.698385Z","shell.execute_reply.started":"2025-12-27T19:09:06.661151Z","shell.execute_reply":"2025-12-27T19:09:49.697747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}