{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":113558,"databundleVersionId":14456136,"sourceType":"competition"},{"sourceId":13686778,"sourceType":"datasetVersion","datasetId":8704744},{"sourceId":275637768,"sourceType":"kernelVersion"},{"sourceId":4534,"sourceType":"modelInstanceVersion","modelInstanceId":3326,"modelId":986}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## Original Notebook https://www.kaggle.com/code/djamilabenchikh/cnn-dinov2-hybrid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:55:50.708172Z","iopub.execute_input":"2025-11-14T21:55:50.708432Z","iopub.status.idle":"2025-11-14T21:55:50.711871Z","shell.execute_reply.started":"2025-11-14T21:55:50.708405Z","shell.execute_reply":"2025-11-14T21:55:50.711054Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📦 Imports","metadata":{}},{"cell_type":"code","source":"import os, cv2, json, math, random\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom PIL import Image\n\nimport torch\nfrom torch import  nn\nimport torch.nn.functional as F\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:55:51.045046Z","iopub.execute_input":"2025-11-14T21:55:51.045282Z","iopub.status.idle":"2025-11-14T21:55:53.108947Z","shell.execute_reply.started":"2025-11-14T21:55:51.045263Z","shell.execute_reply":"2025-11-14T21:55:53.108325Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"$$ $$","metadata":{}},{"cell_type":"markdown","source":"## 🦖 DINO Model","metadata":{}},{"cell_type":"code","source":"BASE_DIR  = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection\"\nAUTH_DIR  = f\"{BASE_DIR}/train_images/authentic\"\nTEST_DIR  = f\"{BASE_DIR}/test_images\"\nDINO_PATH = \"/kaggle/input/dinov2/pytorch/base/1\"\n\nIMG_SIZE = 512\n\nfrom transformers import AutoImageProcessor, AutoModel\nprocessor = AutoImageProcessor.from_pretrained(DINO_PATH, local_files_only=True)\nencoder = AutoModel.from_pretrained(DINO_PATH, local_files_only=True).eval().to(device)\n\nclass DinoTinyDecoder(nn.Module):\n    def __init__(self, in_ch=768, out_ch=1):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Conv2d(in_ch,256,3,padding=1), nn.ReLU(),\n            nn.Conv2d(256,64,3,padding=1), nn.ReLU(),\n            nn.Conv2d(64,out_ch,1)\n        )\n    def forward(self, f, size):\n        return self.net(F.interpolate(f, size=size, mode=\"bilinear\", align_corners=False))\n\nclass DinoSegmenter(nn.Module):\n    def __init__(self, encoder, processor):\n        super().__init__()\n        self.encoder, self.processor = encoder, processor\n        for p in self.encoder.parameters(): p.requires_grad = False\n        self.seg_head = DinoTinyDecoder(768,1)\n        \n    def forward_features(self,x):\n        imgs = (x*255).clamp(0,255).byte().permute(0,2,3,1).cpu().numpy()\n        inputs = self.processor(images=list(imgs), return_tensors=\"pt\").to(x.device)\n        with torch.no_grad(): \n            feats = self.encoder(**inputs).last_hidden_state\n        B,N,C = feats.shape\n        fmap = feats[:,1:,:].permute(0,2,1)\n        s = int(math.sqrt(N-1))\n        fmap = fmap.reshape(B,C,s,s)\n        return fmap\n        \n    def forward_seg(self,x):\n        fmap = self.forward_features(x)\n        return self.seg_head(fmap,(IMG_SIZE,IMG_SIZE))\n\n\n# Load trained model\nMODEL_PATH = '/kaggle/input/cnn-dinov2-hybrid/model_seg_final.pt'\nforgery_model = DinoSegmenter(encoder, processor).to(device)\nforgery_model.load_state_dict(torch.load(MODEL_PATH, map_location=device))\nforgery_model.eval()\nprint(\"✅ Model loaded successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:56:06.653926Z","iopub.execute_input":"2025-11-14T21:56:06.654371Z","iopub.status.idle":"2025-11-14T21:56:37.539591Z","shell.execute_reply.started":"2025-11-14T21:56:06.654346Z","shell.execute_reply":"2025-11-14T21:56:37.538776Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"$$ $$","metadata":{}},{"cell_type":"markdown","source":"## 🧩 Meta SAM ","metadata":{}},{"cell_type":"code","source":"from transformers import pipeline\n# I have put the model files as a dataset because i wasn't able to import it as a model input in kaggle and internet access was off\ngenerator =  pipeline(\"mask-generation\", \n                      model = \"/kaggle/input/sam-vit-base-dataset\", \n                      device = device, \n                      points_per_batch = 64)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:56:42.308515Z","iopub.execute_input":"2025-11-14T21:56:42.308805Z","iopub.status.idle":"2025-11-14T21:56:48.526034Z","shell.execute_reply.started":"2025-11-14T21:56:42.308787Z","shell.execute_reply":"2025-11-14T21:56:48.525451Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"$$ $$","metadata":{}},{"cell_type":"markdown","source":"## ƒ Utility Functions","metadata":{}},{"cell_type":"code","source":"# --- RLE Encoder for Kaggle Submission ---\ndef _rle_encode_jit(x: np.ndarray, fg_val: int = 1) -> list:\n    dots = np.where(x.T.flatten() == fg_val)[0]\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return [int(x) for x in run_lengths]  # convert all to native int\n\ndef rle_encode(masks: list[np.ndarray], fg_val: int = 1) -> str:\n    return ';'.join([json.dumps(_rle_encode_jit(m.astype(np.uint8), fg_val)) for m in masks])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:56:48.911255Z","iopub.execute_input":"2025-11-14T21:56:48.911567Z","iopub.status.idle":"2025-11-14T21:56:48.917463Z","shell.execute_reply.started":"2025-11-14T21:56:48.911544Z","shell.execute_reply":"2025-11-14T21:56:48.916599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enhanced_adaptive_mask(prob, alpha_grad=0.35, alpha_std=0.3):\n    gx = cv2.Sobel(prob, cv2.CV_32F, 1, 0, ksize=3)\n    gy = cv2.Sobel(prob, cv2.CV_32F, 0, 1, ksize=3)\n    grad_mag = np.sqrt(gx**2 + gy**2)\n    grad_norm = grad_mag / (grad_mag.max() + 1e-6)\n    enhanced = (1 - alpha_grad) * prob + alpha_grad * grad_norm\n    enhanced = cv2.GaussianBlur(enhanced, (3,3), 0)\n    thr = np.mean(enhanced) + alpha_std * np.std(enhanced)\n    mask = (enhanced > thr).astype(np.uint8)\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5,5), np.uint8))\n    mask = cv2.morphologyEx(mask, cv2.MORPH_OPEN, np.ones((3,3), np.uint8))\n    return mask, thr\n\ndef pipeline_final(prob, area_threshold = 400, mean_threshold = 0.35, alpha_grad = 0.35, alpha_std=0.3):\n    mask, thr = enhanced_adaptive_mask(prob, alpha_grad, alpha_std)\n    \n    area = int(mask.sum())\n    if area < area_threshold:\n        return \"authentic\" \n        \n    mean_inside = float(prob[mask==1].mean())\n    if mean_inside < mean_threshold:\n        return \"authentic\" \n        \n    return \"forged\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:56:50.030791Z","iopub.execute_input":"2025-11-14T21:56:50.031510Z","iopub.status.idle":"2025-11-14T21:56:50.037610Z","shell.execute_reply.started":"2025-11-14T21:56:50.031485Z","shell.execute_reply":"2025-11-14T21:56:50.036870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"area_threshold = 400\nmean_threshold = 0.35\nalpha_grad = 0.35\nalpha_std=0.3\n\ndef predict_forged(image_path):\n    with torch.no_grad():\n        pil = Image.open(image_path).convert(\"RGB\").resize((IMG_SIZE, IMG_SIZE))\n        pil = np.array(pil, np.float32) / 255.\n        x = torch.from_numpy(pil).permute(2,0,1)[None].to(device)\n        logits = forgery_model.forward_seg(x)\n        prob = torch.sigmoid(logits)[0,0].cpu().numpy()\n    pred = pipeline_final(prob, area_threshold, mean_threshold, alpha_grad, alpha_std)\n    return pred\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:56:56.933133Z","iopub.execute_input":"2025-11-14T21:56:56.933470Z","iopub.status.idle":"2025-11-14T21:56:56.938814Z","shell.execute_reply.started":"2025-11-14T21:56:56.933445Z","shell.execute_reply":"2025-11-14T21:56:56.938067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_image(image_path):\n    pred = predict_forged(image_path)\n    if pred == 'authentic':\n        return pred\n\n    try:\n        # Dummy Segmentation using SAM\n        with torch.no_grad():\n            outputs = generator(image_path, points_per_batch = 32)\n        torch.cuda.empty_cache()\n        sam_masks = [m.astype(np.uint8) for m in outputs[\"masks\"][1:]]\n        if len(sam_masks) > 0:\n            pred_mask = np.zeros_like(sam_masks[0], dtype=np.uint8)\n            for sam_mask in sam_masks:\n                pred_mask = np.logical_or(pred_mask, sam_mask)\n            rle = rle_encode([pred_mask])\n            return rle\n                    \n    except Exception as e:\n        print(f\"[-] Error segmenting image: {image_path}: {e}\")\n        return 'authentic'\n    \n    \n    return 'authentic'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:57:00.611220Z","iopub.execute_input":"2025-11-14T21:57:00.611927Z","iopub.status.idle":"2025-11-14T21:57:00.617330Z","shell.execute_reply.started":"2025-11-14T21:57:00.611900Z","shell.execute_reply":"2025-11-14T21:57:00.616579Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"$$ $$","metadata":{}},{"cell_type":"markdown","source":"## 📊 Prediction","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images'\nsubmission = []\n\nfor f in tqdm(os.listdir(data_dir)):\n    image_path = os.path.join(data_dir, f)\n    if not image_path.endswith(\".png\"):\n        continue\n        \n    id = f.split(\".\")[0]\n    \n    if int(id) == 45:\n        pred = 'authentic'\n        \n    else:\n        pred = predict_image(image_path)\n\n    submission.append({\n        \"case_id\": id,\n        'annotation': pred\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:57:04.699162Z","iopub.execute_input":"2025-11-14T21:57:04.699481Z","iopub.status.idle":"2025-11-14T21:57:04.720203Z","shell.execute_reply.started":"2025-11-14T21:57:04.699460Z","shell.execute_reply":"2025-11-14T21:57:04.719310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame(submission)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-14T21:57:08.476039Z","iopub.execute_input":"2025-11-14T21:57:08.476323Z","iopub.status.idle":"2025-11-14T21:57:08.487873Z","shell.execute_reply.started":"2025-11-14T21:57:08.476302Z","shell.execute_reply":"2025-11-14T21:57:08.487251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}