{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport re\nfrom collections import defaultdict\nimport csv\nimport sys\nimport math\nfrom typing import List, Tuple, Optional\n\n# Change paths if necessary\nROOT = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection\"\nDIR_TRAIN_AUTH = Path(f\"{ROOT}/train_images/authentic\")\nDIR_TRAIN_FORG = Path(f\"{ROOT}/train_images/forged\")\nDIR_TEST       = Path(f\"{ROOT}/test_images\")  # not used, kept for compatibility\nOUTPUT_CSV     = Path(\"/kaggle/working/pairs.csv\")\n\n# Method selection\nUSE_PHASH = True           # Use pHash for unmatched images by name?\nPHASH_THRESHOLD = 10       # Hamming distance threshold (lower is better)\n\nUSE_SSIM = False           # (Optional) Use SSIM for remaining unmatched -after pHash?\nSSIM_THRESHOLD = 0.70      # SSIM threshold (higher is better)\nSSIM_MAX_SIDE = 512        # resize for speed\n\n# If the dataset is too large, you can limit the number of test samples (None = no limit)\nDEBUG_LIMIT_FORGED = None   # e.g. 200\n\n# ------------------- Dependencies -------------------\n# Install required packages (allowed in Kaggle). If already installed, this section will just pass.\ntry:\n    import imagehash  # type: ignore\n    from PIL import Image  # type: ignore\nexcept Exception:\n    !pip -q install imagehash\n    import imagehash\n    from PIL import Image\n\nif USE_SSIM:\n    try:\n        import cv2  # type: ignore\n        from skimage.metrics import structural_similarity as ssim  # type: ignore\n    except Exception:\n        !pip -q install scikit-image opencv-python-headless\n        import cv2\n        from skimage.metrics import structural_similarity as ssim\n\n# ------------------- Utilities -------------------\nIMG_EXTS = {\".png\", \".jpg\", \".jpeg\", \".tif\", \".tiff\", \".bmp\"}\n\ndef list_images(root: Path) -> List[Path]:\n    return sorted([p for p in root.rglob(\"*\") if p.suffix.lower() in IMG_EXTS])\n\ndef normalize_name(p: Path) -> str:\n    s = p.stem.lower()\n    s = s.replace(\"-\", \"_\").replace(\" \", \"_\")\n    # Remove common ending tags\n    s = re.sub(r\"(authentic|auth|original|orig|clean|real|gt)$\", \"\", s)\n    s = re.sub(r\"(forg(ed)?|fake|tampered|edit(ed)?|manipulated?)$\", \"\", s)\n    s = re.sub(r\"(__+|_+$)\", \"\", s)\n    return s\n\ndef safe_open_image(path: Path) -> Optional[Image.Image]:\n    try:\n        return Image.open(path).convert(\"RGB\")\n    except Exception:\n        return None\n\ndef phash_value(path: Path):\n    im = safe_open_image(path)\n    if im is None:\n        return None\n    try:\n        return imagehash.phash(im)\n    except Exception:\n        return None\n\ndef hamming_distance(h1, h2) -> int:\n    return abs(h1 - h2)\n\ndef load_gray_resized_cv2(path: Path, max_side=512):\n    import numpy as np\n    img = cv2.imread(str(path), cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        return None\n    h, w = img.shape[:2]\n    scale = min(1.0, max_side / max(h, w))\n    if scale < 1.0:\n        img = cv2.resize(img, (int(w * scale), int(h * scale)), interpolation=cv2.INTER_AREA)\n    return img\n\ndef best_match_by_phash(gf: Path, auth_hashes: List[Tuple[Path, object]]) -> Tuple[Optional[Path], Optional[int]]:\n    gh = phash_value(gf)\n    if gh is None:\n        return (None, None)\n    best_p, best_d = None, None\n    for ap, ah in auth_hashes:\n        if ah is None:\n            continue\n        d = hamming_distance(gh, ah)\n        if best_d is None or d < best_d:\n            best_p, best_d = ap, d\n    return (best_p, best_d)\n\ndef best_match_by_ssim(gf: Path, auth_imgs: List[Tuple[Path, 'np.ndarray']], max_side=512) -> Tuple[Optional[Path], Optional[float]]:\n    import numpy as np\n    gi = load_gray_resized_cv2(gf, max_side=max_side)\n    if gi is None:\n        return (None, None)\n    best_p, best_sc = None, None\n    gh, gw = gi.shape[:2]\n    for ap, ai in auth_imgs:\n        if ai is None:\n            continue\n        ah, aw = ai.shape[:2]\n        # Simple alignment if dimensions differ\n        if (ah, aw) != (gh, gw):\n            ai_r = cv2.resize(ai, (gw, gh), interpolation=cv2.INTER_AREA)\n        else:\n            ai_r = ai\n        try:\n            sc = ssim(gi, ai_r)\n        except Exception:\n            continue\n        if best_sc is None or sc > best_sc:\n            best_p, best_sc = ap, sc\n    return (best_p, best_sc)\n\n# ------------------- Main Flow -------------------\ndef main():\n    # 1) List image files\n    auth_files = list_images(DIR_TRAIN_AUTH)\n    forg_files = list_images(DIR_TRAIN_FORG)\n    if DEBUG_LIMIT_FORGED is not None:\n        forg_files = forg_files[:DEBUG_LIMIT_FORGED]\n\n    print(f\"#auth = {len(auth_files)}, #forg = {len(forg_files)}\")\n\n    # 2) Pairing based on file names\n    auth_map = defaultdict(list)\n    for p in auth_files:\n        auth_map[normalize_name(p)].append(p)\n\n    pairs: List[Tuple[Path, Path, str, float]] = []\n    unmatched_forg: List[Path] = []\n\n    for gf in forg_files:\n        key = normalize_name(gf)\n        if key in auth_map and len(auth_map[key]) > 0:\n            # If multiple candidates exist, take the first for now\n            ap = auth_map[key][0]\n            pairs.append((ap, gf, \"name\", 0.0))\n        else:\n            unmatched_forg.append(gf)\n\n    print(f\"Name-matched pairs: {len(pairs)}\")\n    print(f\"Unmatched forged by name: {len(unmatched_forg)}\")\n\n    # 3) pHash matching for unmatched images\n    still_unmatched: List[Path] = unmatched_forg\n    if USE_PHASH and len(still_unmatched) > 0:\n        print(\"Computing pHash for authentic images...\")\n        auth_hashes = []\n        for ap in auth_files:\n            try:\n                ah = phash_value(ap)\n            except Exception:\n                ah = None\n            auth_hashes.append((ap, ah))\n\n        print(\"Matching unmatched forged by pHash...\")\n        newly_paired = 0\n        next_unmatched = []\n        for gf in still_unmatched:\n            ap, dist = best_match_by_phash(gf, auth_hashes)\n            if ap is not None and dist is not None and dist <= PHASH_THRESHOLD:\n                pairs.append((ap, gf, \"phash\", float(dist)))\n                newly_paired += 1\n            else:\n                next_unmatched.append(gf)\n        still_unmatched = next_unmatched\n        print(f\"pHash new pairs (<= {PHASH_THRESHOLD}): {newly_paired}\")\n        print(f\"Remaining unmatched after pHash: {len(still_unmatched)}\")\n\n    # 4) Optional SSIM matching\n    if USE_SSIM and len(still_unmatched) > 0:\n        print(\"Preloading grayscale resized authentic images for SSIM...\")\n        auth_imgs = []\n        for ap in auth_files:\n            try:\n                ai = load_gray_resized_cv2(ap, max_side=SSIM_MAX_SIDE)\n            except Exception:\n                ai = None\n            auth_imgs.append((ap, ai))\n\n        print(\"Matching unmatched forged by SSIM...\")\n        newly_paired = 0\n        next_unmatched = []\n        for gf in still_unmatched:\n            try:\n                ap, sc = best_match_by_ssim(gf, auth_imgs, max_side=SSIM_MAX_SIDE)\n            except Exception:\n                ap, sc = (None, None)\n            if ap is not None and sc is not None and sc >= SSIM_THRESHOLD:\n                pairs.append((ap, gf, \"ssim\", float(sc)))\n                newly_paired += 1\n            else:\n                next_unmatched.append(gf)\n        still_unmatched = next_unmatched\n        print(f\"SSIM new pairs (>= {SSIM_THRESHOLD}): {newly_paired}\")\n        print(f\"Remaining unmatched after SSIM: {len(still_unmatched)}\")\n\n    # 5) Save output CSV\n    OUTPUT_CSV.parent.mkdir(parents=True, exist_ok=True)\n    with open(OUTPUT_CSV, \"w\", newline=\"\", encoding=\"utf-8\") as f:\n        writer = csv.writer(f)\n        writer.writerow([\"auth_path\", \"forg_path\", \"method\", \"score\"])\n        for ap, gf, m, s in pairs:\n            writer.writerow([str(ap), str(gf), m, s])\n\n    print(f\"\\nSaved pairs: {len(pairs)} -> {OUTPUT_CSV}\")\n    if len(still_unmatched) > 0:\n        print(\"Sample unmatched forged (up to 10):\")\n        for g in still_unmatched[:10]:\n            print(\" -\", g)\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T13:40:17.831239Z","iopub.execute_input":"2025-11-06T13:40:17.831594Z","iopub.status.idle":"2025-11-06T13:43:26.422794Z","shell.execute_reply.started":"2025-11-06T13:40:17.831569Z","shell.execute_reply":"2025-11-06T13:43:26.421484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom pathlib import Path\n\nPAIRS_CSV = \"/kaggle/working/pairs.csv\"\nIMG_SIZE = 256  \n\ndef _open_rgb(path):\n    img = Image.open(path).convert(\"RGB\")\n    return img.resize((IMG_SIZE, IMG_SIZE), Image.BILINEAR)\n\ndef _abs_diff(a_img, f_img):\n    a = np.array(a_img, dtype=np.int16)\n    f = np.array(f_img, dtype=np.int16)\n    d = np.abs(f - a).astype(np.uint8)\n    return Image.fromarray(d)\n\ndef show_paired_samples(pairs_csv=PAIRS_CSV, n=6, seed=42):\n    df = pd.read_csv(pairs_csv)\n    if len(df) == 0:\n        print(\"pairs.csv is empty.\")\n        return\n\n    np.random.seed(seed)\n    idx = np.random.choice(len(df), size=min(n, len(df)), replace=False)\n    rows = df.iloc[idx].reset_index(drop=True)\n\n    fig, axes = plt.subplots(len(rows), 3, figsize=(12, 4*len(rows)))\n    if len(rows) == 1:\n        axes = np.array([axes])  # ensure 2D\n\n    for r, row in rows.iterrows():\n        a_path, f_path = row[\"auth_path\"], row[\"forg_path\"]\n        method, score = row.get(\"method\", \"name\"), row.get(\"score\", 0.0)\n\n        try:\n            a_img = _open_rgb(a_path)\n            f_img = _open_rgb(f_path)\n        except Exception as e:\n            print(f\"⚠️ Skipping row {r} due to read error: {e}\")\n            continue\n\n        d_img = _abs_diff(a_img, f_img)\n\n        # Authentic\n        axes[r, 0].imshow(a_img)\n        axes[r, 0].set_title(f\"Authentic\\n{Path(a_path).name}\", fontsize=10)\n        axes[r, 0].axis(\"off\")\n\n        # Forged\n        axes[r, 1].imshow(f_img)\n        axes[r, 1].set_title(f\"Forged\\n{Path(f_path).name}\", fontsize=10)\n        axes[r, 1].axis(\"off\")\n\n        # |F-A| with pairing info\n        axes[r, 2].imshow(d_img)\n        axes[r, 2].set_title(f\"|F - A|   (method={method}, score={score})\", fontsize=10)\n        axes[r, 2].axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()\n\n# Run it:\nshow_paired_samples(PAIRS_CSV, n=6, seed=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T13:44:37.557389Z","iopub.execute_input":"2025-11-06T13:44:37.557740Z","iopub.status.idle":"2025-11-06T13:44:40.444146Z","shell.execute_reply.started":"2025-11-06T13:44:37.557715Z","shell.execute_reply":"2025-11-06T13:44:40.442768Z"}},"outputs":[],"execution_count":null}]}