{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":4534,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":3326,"modelId":986}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:36:57.378632Z","iopub.execute_input":"2025-12-20T04:36:57.379346Z","iopub.status.idle":"2025-12-20T04:36:57.460159Z","shell.execute_reply.started":"2025-12-20T04:36:57.379316Z","shell.execute_reply":"2025-12-20T04:36:57.459294Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PROJECT: Optima-V (Version 1.0)\nCOMPETITION: Recodai-LUC Scientific Image Forgery Detection\n\nDESCRIPTION:\nThis notebook implements an automated forgery detection pipeline using DINOv2 (Vision Transformer) \nas a frozen feature extractor coupled with a lightweight convolutional decoder. \n\nSTRATEGY:\n1. Feature Extraction: Utilizing DINOv2-Base (Frozen) for robust patch-level representations.\n2. Inference: 3-Way Test Time Augmentation (TTA) - Original, Horizontal, and Vertical Flips.\n3. Post-Processing: Morphology Open operations and Area Thresholding (min 500px) to filter noise.\n4. Robustness: GPU-accelerated normalization and fast RLE encoding for submission.\n\n# Version 2\n\n-Make more EDA to explain\n\nKey Features:\n\n- Backbone: DINOv2 (Frozen) for robust self-supervised feature extraction.\n- TTA (Test Time Augmentation): 3-way flips (Horizontal/Vertical) to stabilize mask predictions.\n- Morphological Post-processing: Using MORPH_OPEN to remove salt-and-pepper noise.\n- Area Filtering: A min_area_threshold of 500px to filter out insignificant candidates.\n\nGOAL: Establish a high-performance baseline for scientific image integrity.","metadata":{}},{"cell_type":"markdown","source":"# Analytic/processing and submission","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom transformers import AutoModel\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nimport json\nimport math\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\n\n# --- CONFIGURATION ---\nclass CFG:\n    test_dir = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images\"\n    sample_sub = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv\"\n    dino_path = \"/kaggle/input/dinov2/pytorch/base/1\" # Local Kaggle path\n    \n    img_size = 512\n    batch_size = 8\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    \n    # Decision Thresholds\n    mask_threshold = 0.5\n    min_area_threshold = 500  # Ignore \"forgeries\" smaller than 500 pixels\n\n# --- MODEL DEFINITION ---\n\nclass DinoTinyDecoder(nn.Module):\n    def __init__(self, in_ch=768, out_ch=1):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Conv2d(in_ch, 256, 3, padding=1), nn.ReLU(),\n            nn.Conv2d(256, 64, 3, padding=1), nn.ReLU(),\n            nn.Conv2d(64, out_ch, 1)\n        )\n\n    def forward(self, f, size):\n        return self.net(F.interpolate(f, size=size, mode=\"bilinear\", align_corners=False))\n\nclass DinoSegmenter(nn.Module):\n    def __init__(self, model_path):\n        super().__init__()\n        # Load local backbone\n        self.encoder = AutoModel.from_pretrained(model_path)\n        for p in self.encoder.parameters():\n            p.requires_grad = False\n            \n        self.seg_head = DinoTinyDecoder(768, 1)\n        \n        # GPU Normalization constants (ImageNet)\n        self.register_buffer(\"mean\", torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1))\n        self.register_buffer(\"std\", torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1))\n\n    def forward(self, x):\n        # 1. GPU Preprocessing\n        x = (x - self.mean) / self.std\n        \n        # 2. Features\n        feats = self.encoder(x).last_hidden_state\n        B, N, C = feats.shape\n        s = int(math.sqrt(N - 1))\n        fmap = feats[:, 1:, :].permute(0, 2, 1).reshape(B, C, s, s)\n        \n        # 3. Mask\n        return self.seg_head(fmap, (CFG.img_size, CFG.img_size))\n\n# --- DATASET & UTILS ---\n\nclass ForgeryTestDataset(Dataset):\n    def __init__(self, img_dir, img_size):\n        self.img_paths = sorted([os.path.join(img_dir, f) for f in os.listdir(img_dir)])\n        self.img_size = img_size\n\n    def __len__(self):\n        return len(self.img_paths)\n\n    def __getitem__(self, idx):\n        path = self.img_paths[idx]\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        orig_h, orig_w = img.shape[:2]\n        \n        img_resized = cv2.resize(img, (self.img_size, self.img_size))\n        img_tensor = torch.from_numpy(img_resized).permute(2, 0, 1).float() / 255.0\n        \n        return img_tensor, Path(path).stem, (orig_w, orig_h)\n\ndef rle_encode_fast(mask):\n    pixels = mask.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return json.dumps(runs.tolist()) if len(runs) > 0 else \"authentic\"\n\n# --- INFERENCE LOOP ---\n\n# Initialize\nmodel = DinoSegmenter(CFG.dino_path).to(CFG.device)\nmodel.eval()\n\ntest_ds = ForgeryTestDataset(CFG.test_dir, CFG.img_size)\nloader = DataLoader(test_ds, batch_size=CFG.batch_size, num_workers=2, shuffle=False)\n\nresults = []\n\n\n\nwith torch.no_grad():\n    for imgs, case_ids, (orig_ws, orig_hs) in tqdm(loader, desc=\"Inference\"):\n        imgs = imgs.to(CFG.device)\n        \n        # 3-Way TTA (Original + Horizontal + Vertical)\n        out1 = torch.sigmoid(model(imgs))\n        out2 = torch.sigmoid(model(torch.flip(imgs, dims=[3])))\n        out3 = torch.sigmoid(model(torch.flip(imgs, dims=[2])))\n        \n        # Average the predictions\n        preds = (out1 + torch.flip(out2, dims=[3]) + torch.flip(out3, dims=[2])) / 3\n        preds_np = preds.squeeze(1).cpu().numpy()\n        \n        for i in range(len(preds_np)):\n            p = preds_np[i]\n            \n            # 1. Initial Mask\n            mask = (p > CFG.mask_threshold).astype(np.uint8)\n            \n            # 2. Clean up noise (Morphology)\n            # This removes tiny specks that aren't real forgeries\n            kernel = np.ones((3,3), np.uint8)\n            mask = cv2.morphologyEx(mask, cv2.MORPH_OPEN, kernel)\n            \n            # 3. Decision Logic\n            if mask.sum() < CFG.min_area_threshold:\n                annotation = \"authentic\"\n            else:\n                orig_size = (int(orig_ws[i]), int(orig_hs[i]))\n                mask_full = cv2.resize(mask, orig_size, interpolation=cv2.INTER_NEAREST)\n                annotation = rle_encode_fast(mask_full)\n            \n            results.append({\"case_id\": str(case_ids[i]), \"annotation\": annotation})\n# --- SUBMISSION ---\n\npred_df = pd.DataFrame(results)\n\n# Load the sample submission\nsub_df = pd.read_csv(CFG.sample_sub)\n\n# CRITICAL FIX: Ensure both columns are strings to avoid the Merge Error\npred_df['case_id'] = pred_df['case_id'].astype(str)\nsub_df['case_id'] = sub_df['case_id'].astype(str)\n\n# Perform the merge\nfinal_sub = sub_df[['case_id']].merge(pred_df, on='case_id', how='left')\n\n# Fill missing values (if any test images weren't processed) with \"authentic\"\nfinal_sub['annotation'] = final_sub['annotation'].fillna(\"authentic\")\n\n# Save to csv\nfinal_sub.to_csv(\"submission.csv\", index=False)\n\nprint(f\"Successfully saved {len(final_sub)} predictions to submission.csv\")\nfinal_sub.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:36:57.461763Z","iopub.execute_input":"2025-12-20T04:36:57.462054Z","iopub.status.idle":"2025-12-20T04:36:58.842358Z","shell.execute_reply.started":"2025-12-20T04:36:57.462030Z","shell.execute_reply":"2025-12-20T04:36:58.841703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# Add this at the very end of your notebook\nprint(\"-\" * 30)\nprint(f\"Inference Complete!\")\nprint(f\"Total Images Processed: {len(final_sub)}\")\nprint(f\"Forgeries Detected: {len(final_sub[final_sub['annotation'] != 'authentic'])}\")\nprint(f\"Authentic Images: {len(final_sub[final_sub['annotation'] == 'authentic'])}\")\nprint(\"-\" * 30)\n\n# This forces the dataframe to display in the output logs\ndisplay(final_sub.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:36:58.843556Z","iopub.execute_input":"2025-12-20T04:36:58.843774Z","iopub.status.idle":"2025-12-20T04:36:58.854234Z","shell.execute_reply.started":"2025-12-20T04:36:58.843751Z","shell.execute_reply":"2025-12-20T04:36:58.853377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\ndef visualize_dino_pca(img_path, model):\n    model.eval()\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    h, w = img.shape[:2]\n    \n    # 1. Get features from the encoder\n    img_input = cv2.resize(img, (CFG.img_size, CFG.img_size))\n    img_input = torch.from_numpy(img_input).permute(2, 0, 1).float().unsqueeze(0).to(CFG.device) / 255.0\n    \n    with torch.no_grad():\n        # Pre-process same as model\n        x = (img_input - model.mean) / model.std\n        feats = model.encoder(x).last_hidden_state # [1, N, 768]\n        \n    # 2. Extract patch tokens (exclude CLS token)\n    features = feats[0, 1:, :].cpu().numpy() # [N-1, 768]\n    \n    # 3. Reduce to 3 components for RGB visualization\n    pca = PCA(n_components=3)\n    pca_features = pca.fit_transform(features) # [N-1, 3]\n    \n    # 4. Reshape back to grid (e.g., 36x36 for 512px input)\n    s = int(math.sqrt(features.shape[0]))\n    pca_img = pca_features.reshape(s, s, 3)\n    \n    # 5. Normalize for display\n    pca_img = (pca_img - pca_img.min()) / (pca_img.max() - pca_img.min())\n    pca_img = cv2.resize(pca_img, (w, h), interpolation=cv2.INTER_NEAREST)\n\n    # Plot\n    plt.figure(figsize=(12, 6))\n    plt.subplot(1, 2, 1)\n    plt.imshow(img)\n    plt.title(\"Original Scientific Image\")\n    plt.axis('off')\n    \n    plt.subplot(1, 2, 2)\n    plt.imshow(pca_img)\n    plt.title(\"DINOv2 Feature PCA (Texture Signature)\")\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:36:58.856073Z","iopub.execute_input":"2025-12-20T04:36:58.856304Z","iopub.status.idle":"2025-12-20T04:36:58.870296Z","shell.execute_reply.started":"2025-12-20T04:36:58.856280Z","shell.execute_reply":"2025-12-20T04:36:58.869568Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"PCA Visualization","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef visualize_prediction(img_path, model, device):\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    # Preprocess\n    img_t = cv2.resize(img, (CFG.img_size, CFG.img_size))\n    img_t = torch.from_numpy(img_t).permute(2, 0, 1).float().unsqueeze(0).to(device) / 255.0\n    \n    # Predict\n    with torch.no_grad():\n        pred = torch.sigmoid(model(img_t)).cpu().numpy()[0, 0]\n    \n    # Plot\n    fig, ax = plt.subplots(1, 3, figsize=(15, 5))\n    ax[0].imshow(img)\n    ax[0].set_title(\"Original Image\")\n    \n    ax[1].imshow(pred, cmap='jet')\n    ax[1].set_title(\"Forgery Heatmap (DINOv2)\")\n    \n    mask = (pred > CFG.mask_threshold).astype(np.uint8)\n    ax[2].imshow(img)\n    ax[2].imshow(mask, alpha=0.4, cmap='Reds') # Overlay\n    ax[2].set_title(\"Predicted Forgery Overlay\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:36:58.871197Z","iopub.execute_input":"2025-12-20T04:36:58.871500Z","iopub.status.idle":"2025-12-20T04:36:58.889921Z","shell.execute_reply.started":"2025-12-20T04:36:58.871462Z","shell.execute_reply":"2025-12-20T04:36:58.889230Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Visualizing Predictions","metadata":{}},{"cell_type":"code","source":"# Pick 3 random images from the test set to show off\ntest_samples = test_ds.img_paths[:3] \n\nfor path in test_samples:\n    visualize_prediction(path, model, CFG.device) # Using the function I gave you earlier\n    visualize_dino_pca(path, model) # Showing the \"Internal Logic\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:36:58.890784Z","iopub.execute_input":"2025-12-20T04:36:58.891020Z","iopub.status.idle":"2025-12-20T04:37:00.386940Z","shell.execute_reply.started":"2025-12-20T04:36:58.891000Z","shell.execute_reply":"2025-12-20T04:37:00.386134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"EDA Plot: Error Level Analysis (ELA)","metadata":{}},{"cell_type":"code","source":"def plot_ela(img_path, quality=90):\n    \"\"\"\n    Shows the difference between the original image and a re-compressed version.\n    Modified areas (forgeries) usually show higher error levels.\n    \"\"\"\n    original = cv2.imread(img_path)\n    original = cv2.cvtColor(original, cv2.COLOR_BGR2RGB)\n    \n    # Save and reload at a specific quality to find compression differences\n    cv2.imwrite(\"temp.jpg\", cv2.cvtColor(original, cv2.COLOR_RGB2BGR), [cv2.IMWRITE_JPEG_QUALITY, quality])\n    compressed = cv2.imread(\"temp.jpg\")\n    compressed = cv2.cvtColor(compressed, cv2.COLOR_BGR2RGB)\n    \n    # Calculate the absolute difference (the \"Error\")\n    ela_map = cv2.absdiff(original, compressed)\n    \n    # Boost the contrast so we can see it\n    max_diff = np.max(ela_map)\n    if max_diff == 0: max_diff = 1\n    scale = 255.0 / max_diff\n    ela_map = (ela_map * scale).astype(np.uint8)\n    \n    return ela_map\n\n# Visualize it for your EDA\nsample_img = CFG.test_dir + \"/\" + os.listdir(CFG.test_dir)[0]\nela_result = plot_ela(sample_img)\n\nplt.figure(figsize=(15, 5))\nplt.subplot(1, 2, 1); plt.imshow(cv2.imread(sample_img)[:,:,::-1]); plt.title(\"Original Image\")\nplt.subplot(1, 2, 2); plt.imshow(ela_result); plt.title(\"ELA (Compression Artifacts)\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T04:37:00.388062Z","iopub.execute_input":"2025-12-20T04:37:00.388688Z","iopub.status.idle":"2025-12-20T04:37:01.043867Z","shell.execute_reply.started":"2025-12-20T04:37:00.388661Z","shell.execute_reply":"2025-12-20T04:37:01.043167Z"}},"outputs":[],"execution_count":null}]}