{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"sourceType":"competition"},{"sourceId":14372983,"sourceType":"datasetVersion","datasetId":9153851},{"sourceId":4534,"sourceType":"modelInstanceVersion","modelInstanceId":3326,"modelId":986}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":63.129007,"end_time":"2025-12-30T04:03:10.929134","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-12-30T04:02:07.800127","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport math\nimport random\nimport torch\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom PIL import Image\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom transformers import AutoImageProcessor, AutoModel\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reproducability","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True \n    torch.backends.cudnn.benchmark = False\n\nseed_everything(42)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Paths, Device, Hyperparmeters","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nBASE_DIR  = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection\"\nDINO_PATH = \"/kaggle/input/dinov2/pytorch/base/1\"\nMODEL_LOC = '/kaggle/input/cnndinov2-pbd/CNNDINOv2-R69/model_seg_final.pt'\n\nIMG_SIZE = 518\n\nAREA_THR = 145        # Compromise between 150 (safe) and 135 (risky)\nP95_THR = 0.55        # tune 0.50–0.65\n\n# Border hyperparams\nBORDER = 6\nBORDER_CC_MEAN_THR = 0.26   # start 0.24–0.30\nBORDER_CC_AREA_THR = 20     # optional: minimum area if it touches border\n\n# Uncertainty gating hyperparams\nUNC_THR = 0.0025        # mean variance threshold (start 0.002–0.004)\nUNC_PENALTY = 0.03      # how much to raise P95_THR when uncertain\nUNC_AREA_PENALTY = 0   # how much to raise AREA_THR when uncertain\n\n# Hysterisis quantiles\nQ_LOW  = 0.92\nQ_HIGH = 0.965","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Definition","metadata":{}},{"cell_type":"code","source":"class DinoTinyDecoder(nn.Module):\n    def __init__(self, in_ch=768, out_ch=1):\n        super().__init__()\n        self.block1 = nn.Sequential(nn.Conv2d(in_ch, 384, 3, 1, 1), nn.ReLU(True), nn.Dropout2d(0.1))\n        self.block2 = nn.Sequential(nn.Conv2d(384, 192, 3, 1, 1), nn.ReLU(True), nn.Dropout2d(0.1))\n        self.block3 = nn.Sequential(nn.Conv2d(192, 96, 3, 1, 1), nn.ReLU(True))\n        self.conv_out = nn.Conv2d(96, out_ch, 1)\n\n    def forward(self, f, target_size):\n        x = F.interpolate(self.block1(f), size=(74, 74), mode='bilinear', align_corners=False)\n        x = F.interpolate(self.block2(x), size=(148, 148), mode='bilinear', align_corners=False)\n        x = F.interpolate(self.block3(x), size=(296, 296), mode='bilinear', align_corners=False)\n        x = self.conv_out(x)\n        return F.interpolate(x, size=target_size, mode='bilinear', align_corners=False)\n\nclass DinoSegmenter(nn.Module):\n    def __init__(self, encoder, processor):\n        super().__init__()\n        self.encoder, self.processor = encoder, processor\n        for p in self.encoder.parameters(): p.requires_grad = False\n        self.seg_head = DinoTinyDecoder(768, 1)\n        \n    def forward_features(self, x):\n        imgs = (x*255).clamp(0,255).byte().permute(0,2,3,1).cpu().numpy()\n        inputs = self.processor(images=list(imgs), return_tensors=\"pt\").to(x.device)\n        feats = self.encoder(**inputs).last_hidden_state\n        B, N, C = feats.shape\n        s = int(math.sqrt(N-1))\n        return feats[:,1:,:].permute(0,2,1).reshape(B, C, s, s)\n\n    def forward_seg(self, x):\n        fmap = self.forward_features(x)\n        return self.seg_head(fmap, (IMG_SIZE, IMG_SIZE))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Processor / Encoder / Weights","metadata":{}},{"cell_type":"code","source":"print(\"⏳ Loading Model...\")\nprocessor = AutoImageProcessor.from_pretrained(DINO_PATH, local_files_only=True, use_fast=False)\nencoder = AutoModel.from_pretrained(DINO_PATH, local_files_only=True).eval().to(device)\nmodel_seg = DinoSegmenter(encoder, processor).to(device)\n\nif os.path.exists(MODEL_LOC):\n    model_seg.load_state_dict(torch.load(MODEL_LOC, map_location=device))\n    print(f\"✅ Loaded weights from: {MODEL_LOC}\")\n\nmodel_seg.eval()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Inference: Multi-scale TTA + Uncertainty","metadata":{}},{"cell_type":"code","source":"@torch.no_grad()\ndef segment_prob_map_multiscale(pil):\n    \"\"\"Run inference at 1.0x (Original & Flip) and 0.75x (Zoom Out).\"\"\"\n    img_tensor = torch.from_numpy(np.array(pil.resize((IMG_SIZE, IMG_SIZE)), np.float32)/255.).permute(2,0,1)[None].to(device)\n    \n    # 1. Scale 1.0x Original\n    prob_1 = torch.sigmoid(model_seg.forward_seg(img_tensor))[0,0]\n    \n    # 2. Scale 1.0x Flip\n    img_flip = torch.flip(img_tensor, [3])\n    prob_1_flip = torch.sigmoid(model_seg.forward_seg(img_flip))[0,0]\n    prob_1_flip = torch.flip(prob_1_flip, [1]) \n    \n    # 3. Scale 0.75x\n    size_small = (392, 392)\n    img_small = F.interpolate(img_tensor, size=size_small, mode='bilinear', align_corners=False)\n    prob_small = torch.sigmoid(model_seg.forward_seg(img_small))\n    prob_small = F.interpolate(prob_small, size=(IMG_SIZE, IMG_SIZE), mode='bilinear', align_corners=False)[0,0]\n\n    # --- Adaptive weights based on agreement ---\n    p1  = prob_1.cpu().numpy()\n    pf  = prob_1_flip.cpu().numpy()\n    ps  = prob_small.cpu().numpy()\n    \n    d_flip  = np.mean(np.abs(p1 - pf))\n    d_small = np.mean(np.abs(p1 - ps))\n    \n    # Convert disagreement to weights (lower disagreement => higher weight)\n    # These numbers are safe starting points.\n    w1 = 0.60\n    wf = 0.20\n    ws = 0.20\n    \n    # If small-scale disagrees a lot, downweight it; if it agrees, upweight it.\n    if d_small > 0.06:\n        ws = 0.10\n        w1 = 0.70\n        wf = 0.20\n    elif d_small < 0.03:\n        ws = 0.30\n        w1 = 0.50\n        wf = 0.20\n    \n    # If flip disagrees a lot, downweight it too\n    if d_flip > 0.06:\n        wf = 0.10\n        w1 = min(0.80, w1 + 0.10)  # give weight back to main\n        ws = 1.0 - w1 - wf\n    \n    # Normalize (just in case)\n    s = w1 + wf + ws\n    w1, wf, ws = w1/s, wf/s, ws/s\n\n    stack = np.stack([p1, pf, ps], axis=0)      # (3, 518, 518)\n    var_map = stack.var(axis=0)                 # (518, 518)\n    unc = float(var_map.mean())                 # scalar uncertainty\n    \n    prob_avg = (p1 * w1) + (pf * wf) + (ps * ws)\n    \n    return prob_avg, unc","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Mask Creation: Edge-Enhanced Hysteresis","metadata":{}},{"cell_type":"code","source":"def enhanced_adaptive_mask(prob, alpha_grad=0.35):\n    # Edge enhancement\n    gx = cv2.Sobel(prob, cv2.CV_32F, 1, 0, ksize=3)\n    gy = cv2.Sobel(prob, cv2.CV_32F, 0, 1, ksize=3)\n    grad_norm = np.sqrt(gx**2 + gy**2)\n    grad_norm = grad_norm / (grad_norm.max() + 1e-6)\n    \n    enhanced = (1 - alpha_grad) * prob + alpha_grad * grad_norm\n    enhanced = cv2.GaussianBlur(enhanced, (3,3), 0)\n    \n    # Hysteresis thresholds (start values)\n    t_high = np.quantile(enhanced, Q_HIGH)\n    t_low  = np.quantile(enhanced, Q_LOW)\n    \n    strong = (enhanced >= t_high).astype(np.uint8)\n    weak   = (enhanced >= t_low).astype(np.uint8)\n    \n    # Keep weak pixels only if connected to strong pixels\n    num, labels, stats, _ = cv2.connectedComponentsWithStats(weak, connectivity=8)\n    mask = np.zeros_like(weak)\n    \n    for i in range(1, num):\n        cc = (labels == i)\n        if (strong[cc].sum() > 0):   # component contains at least one strong pixel\n            mask[cc] = 1\n    \n    thr = (t_low, t_high)  # for debugging\n\n    \n    # REVERT: Use (5,5) kernel. (3,3) was too fragmented.\n    # (5,5) closes gaps better, improving IoU on solid objects.\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5,5), np.uint8))\n    \n    return mask, thr","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Post-processing: Connected Components + Border Rules","metadata":{}},{"cell_type":"code","source":"def filter_connected_components(\n    mask, prob,\n    min_cc_area=30,\n    min_cc_mean=0.22,\n    top_k=None,\n    border=6,\n    border_mean_thr=0.26,\n    border_area_thr=0\n):\n    \"\"\"\n    Keeps components that are large/strong enough.\n    Additionally, components touching the border are kept only if stronger.\n    \"\"\"\n    mask = mask.astype(np.uint8)\n    num, labels, stats, _ = cv2.connectedComponentsWithStats(mask, connectivity=8)\n\n    if num <= 1:\n        return mask\n\n    H, W = mask.shape\n    kept = []\n\n    for i in range(1, num):\n        x = stats[i, cv2.CC_STAT_LEFT]\n        y = stats[i, cv2.CC_STAT_TOP]\n        w = stats[i, cv2.CC_STAT_WIDTH]\n        h = stats[i, cv2.CC_STAT_HEIGHT]\n        area_i = stats[i, cv2.CC_STAT_AREA]\n\n        if area_i < min_cc_area:\n            continue\n\n        cc = (labels == i)\n        mean_i = float(prob[cc].mean())\n        if mean_i < min_cc_mean:\n            continue\n\n        # Does this component touch the border band?\n        touches_border = (\n            x < border or y < border or (x + w) > (W - border) or (y + h) > (H - border)\n        )\n\n        # If it touches border, require higher confidence (and optionally some area)\n        if touches_border:\n            if mean_i < border_mean_thr:\n                continue\n            if border_area_thr > 0 and area_i < border_area_thr:\n                continue\n\n        mass_i = area_i * mean_i\n        kept.append((mass_i, i))\n\n    if not kept:\n        return np.zeros_like(mask)\n\n    kept.sort(reverse=True)\n    if top_k is not None:\n        kept = kept[:top_k]\n\n    out = np.zeros_like(mask)\n    for _, i in kept:\n        out[labels == i] = 1\n\n    return out","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Decision Logic: Final Pipeline","metadata":{}},{"cell_type":"code","source":"def pipeline_final(pil):\n    prob, unc = segment_prob_map_multiscale(pil)\n\n    mask518, thr = enhanced_adaptive_mask(prob)\n    mask518 = filter_connected_components(\n        mask518, prob,\n        min_cc_area=25,\n        min_cc_mean=0.22,\n        top_k=3,\n        border=BORDER,\n        border_mean_thr=BORDER_CC_MEAN_THR,\n        border_area_thr=BORDER_CC_AREA_THR\n    )\n\n\n    area518 = int(mask518.sum())\n\n    if area518 > 0:\n        vals = prob[mask518 == 1]\n        p95_inside = float(np.quantile(vals, 0.95))\n        max_conf = float(vals.max())\n    else:\n        p95_inside = 0.0\n        max_conf = 0.0\n\n    dbg = {\"area\": area518, \"p95\": p95_inside, \"t_low\": float(thr[0]),\n           \"t_high\": float(thr[1]),\"max\": max_conf, \"unc\": unc}\n    # Small but strong gate\n    if area518 < 300:\n        if max_conf < 0.75 or p95_inside < 0.60:\n            return \"authentic\", None, dbg\n\n    # High confidence override\n    if max_conf > 0.95 and area518 > 50:\n        mask_orig = cv2.resize(mask518, pil.size, interpolation=cv2.INTER_NEAREST)\n        return \"forged\", mask_orig, dbg\n\n    # Uncertainty gating\n    p95_thr_eff = P95_THR\n    area_thr_eff = AREA_THR\n    if unc > UNC_THR:\n        p95_thr_eff = P95_THR + UNC_PENALTY\n        area_thr_eff = AREA_THR + UNC_AREA_PENALTY\n\n    # Standard filter with Effective thresholds\n    if area518 < area_thr_eff or p95_inside < p95_thr_eff:\n        return \"authentic\", None, dbg\n        \n    mask_orig = cv2.resize(mask518, pil.size, interpolation=cv2.INTER_NEAREST)\n    return \"forged\", mask_orig, dbg","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RLE Encode","metadata":{}},{"cell_type":"code","source":"def rle_encode(mask):\n    pixels = mask.T.flatten()\n    dots = np.where(pixels == 1)[0]\n    if len(dots) == 0: return \"authentic\"\n    run_lengths, prev = [], -2\n    for b in dots:\n        if b > prev + 1: run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return json.dumps([int(x) for x in run_lengths])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Generating Submission","metadata":{}},{"cell_type":"code","source":"TEST_DIR = f\"{BASE_DIR}/test_images\"\nSAMPLE_SUB = f\"{BASE_DIR}/sample_submission.csv\"\nOUT_PATH = \"submission.csv\"\n\nrows = []\ntest_files = sorted(os.listdir(TEST_DIR))\nprint(f\"🚀 Recovery Run on {len(test_files)} images...\")\n\nfor f in tqdm(test_files):\n    pil = Image.open(Path(TEST_DIR)/f).convert(\"RGB\")\n    label, mask, dbg = pipeline_final(pil)\n    \n    if label == \"authentic\" or mask is None:\n        annot = \"authentic\"\n    else:\n        annot = rle_encode((mask > 0).astype(np.uint8))\n        \n    rows.append({\"case_id\": Path(f).stem, \"annotation\": annot})\n\nsub = pd.DataFrame(rows)\nss = pd.read_csv(SAMPLE_SUB)\nss[\"case_id\"] = ss[\"case_id\"].astype(str)\nsub[\"case_id\"] = sub[\"case_id\"].astype(str)\n\nfinal = ss[[\"case_id\"]].merge(sub, on=\"case_id\", how=\"left\")\nfinal[\"annotation\"] = final[\"annotation\"].fillna(\"authentic\")\nfinal[[\"case_id\", \"annotation\"]].to_csv(OUT_PATH, index=False)\n\nprint(f\"\\n✅ Recovery Submission Saved: {OUT_PATH}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}