{"cells": [{"cell_type": "markdown", "metadata": {}, "source": "# Recod.ai/LUC Scientific Image Forgery Detection - WayneIA V6\n\n**Competition**: recodai-luc-scientific-image-forgery-detection  \n**Prize**: $55,000  \n**Approach**: Aggressive RLE Mask Detection  \n\nCODE_KEY[166] FORGERY_CLASSIFIER_MATRIX\n\nV6: AGGRESSIVE RLE mask output (not just \"forged\" classification)\n\n**Format**: \n- `authentic` = no forgery detected\n- `[start1 len1 start2 len2 ...]` = RLE-encoded binary mask\n\nWayneIA Position_1 OpusPlan | December 23, 2025 | Year-8 RHINOCEROS G9"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# WayneIA V6 - RLE Mask Generation Kernel\n# Competition-compliant: outputs \"authentic\" OR RLE mask string\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom PIL import Image\nfrom scipy.ndimage import uniform_filter\n\nprint(\"=\"*60)\nprint(\"Recod.ai/LUC Forgery Detection - WayneIA V6\")\nprint(\"CODE_KEY[166] FORGERY_CLASSIFIER_MATRIX\")\nprint(\"AGGRESSIVE RLE MASK OUTPUT\")\nprint(\"=\"*60)\n\n# Environment detection\nIN_KAGGLE = os.path.exists('/kaggle/input')\nprint(f\"Kaggle environment: {IN_KAGGLE}\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# RLE Encoding Function (Competition Format)\ndef rle_encode(mask: np.ndarray) -> str:\n    \"\"\"\n    Run-length encode a binary mask.\n    Returns: RLE string like \"[start1 length1 start2 length2 ...]\"\n    \"\"\"\n    # Flatten the mask\n    pixels = mask.flatten()\n    \n    # Find where values change\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0]\n    runs = runs.reshape(-1, 2)\n    \n    # Convert to start, length format\n    rle_pairs = []\n    for start, end in runs:\n        rle_pairs.extend([start, end - start])\n    \n    if len(rle_pairs) == 0:\n        return \"authentic\"\n    \n    return f\"[{' '.join(map(str, rle_pairs))}]\"\n\nprint(\"RLE encoding function loaded\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# Forgery Detection Function\ndef detect_forgery(image_path: Path, threshold_percentile: float = 95.0) -> dict:\n    \"\"\"\n    Detect potential forgery regions using edge + variance analysis.\n    Returns dict with mask, RLE string, and metrics.\n    \"\"\"\n    img = Image.open(image_path).convert('RGB')\n    img_array = np.array(img)\n    \n    # Convert to grayscale for analysis\n    gray = np.mean(img_array, axis=2)\n    \n    # Simple edge detection (Sobel-like)\n    gx = np.abs(np.diff(gray, axis=1, prepend=gray[:, :1]))\n    gy = np.abs(np.diff(gray, axis=0, prepend=gray[:1, :]))\n    edges = np.sqrt(gx**2 + gy**2)\n    \n    # Normalize and threshold\n    edges = edges / edges.max() if edges.max() > 0 else edges\n    \n    # Look for high-variance regions (potential copy-move artifacts)\n    local_mean = uniform_filter(gray, size=15)\n    local_sqr_mean = uniform_filter(gray**2, size=15)\n    local_var = local_sqr_mean - local_mean**2\n    \n    # Normalize variance\n    var_norm = local_var / (local_var.max() + 1e-8)\n    \n    # Combine edge and variance features\n    combined = 0.5 * edges + 0.5 * var_norm\n    \n    # Threshold to create binary mask (AGGRESSIVE: top 5%)\n    threshold = np.percentile(combined, threshold_percentile)\n    mask = (combined > threshold).astype(np.uint8)\n    \n    # Calculate forgery ratio\n    forgery_ratio = mask.sum() / mask.size\n    is_forged = forgery_ratio > 0.01  # If >1% of image is anomalous\n    \n    # Generate RLE\n    if is_forged:\n        rle = rle_encode(mask)\n    else:\n        rle = \"authentic\"\n    \n    return {\n        \"mask\": mask,\n        \"rle\": rle,\n        \"forgery_ratio\": forgery_ratio,\n        \"is_forged\": is_forged,\n        \"shape\": mask.shape\n    }\n\nprint(\"Forgery detection function loaded\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# Configuration\nif IN_KAGGLE:\n    DATA_DIR = Path(\"/kaggle/input/recodai-luc-scientific-image-forgery-detection\")\n    OUTPUT_DIR = Path(\"/kaggle/working\")\nelse:\n    DATA_DIR = Path(\"/mnt/wayne/competitions/recod_ai_luc/extracted\")\n    OUTPUT_DIR = Path(\"/mnt/wayne/competitions/recod_ai_luc/output\")\n\n# List test images\ntest_images_dir = DATA_DIR / \"test_images\"\nif test_images_dir.exists():\n    test_images = sorted(list(test_images_dir.glob(\"*.png\")) + list(test_images_dir.glob(\"*.jpg\")))\n    print(f\"Found {len(test_images)} test images:\")\n    for img in test_images:\n        print(f\"  - {img.name}\")\nelse:\n    print(f\"Test images directory not found: {test_images_dir}\")\n    test_images = []"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# Generate Predictions with RLE Masks\npredictions = []\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Running AGGRESSIVE Forgery Detection\")\nprint(\"=\"*60 + \"\\n\")\n\nfor img_path in test_images:\n    case_id = img_path.stem\n    print(f\"Processing: {img_path.name}\")\n    \n    # Run forgery detection\n    result = detect_forgery(img_path, threshold_percentile=95.0)\n    \n    print(f\"  Shape: {result['shape']}\")\n    print(f\"  Forgery ratio: {result['forgery_ratio']:.4f}\")\n    print(f\"  Is forged: {result['is_forged']}\")\n    \n    # Get RLE annotation\n    annotation = result['rle']\n    \n    if annotation == \"authentic\":\n        print(f\"  Annotation: authentic\")\n    else:\n        print(f\"  Annotation: RLE mask ({len(annotation)} chars)\")\n        # Show first 100 chars of RLE\n        print(f\"  RLE preview: {annotation[:100]}...\")\n    \n    predictions.append({'case_id': case_id, 'annotation': annotation})\n    print()\n\nprint(f\"Total predictions: {len(predictions)}\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# Save Submission\nsubmission_df = pd.DataFrame(predictions)\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nsubmission_path = OUTPUT_DIR / \"submission.csv\"\nsubmission_df.to_csv(submission_path, index=False)\n\nprint(f\"\\nSubmission saved to: {submission_path}\")\nprint(f\"Submission size: {submission_path.stat().st_size} bytes\")\n\n# Show submission summary\nprint(\"\\nSubmission Summary:\")\nprint(\"-\" * 40)\nfor idx, row in submission_df.iterrows():\n    if row['annotation'] == 'authentic':\n        print(f\"  {row['case_id']}: authentic\")\n    else:\n        print(f\"  {row['case_id']}: RLE mask ({len(row['annotation'])} chars)\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "print(\"\\n\" + \"=\"*60)\nprint(\"Recod.ai/LUC WayneIA V6 - COMPLETE\")\nprint(\"CODE_KEY[166] FORGERY_CLASSIFIER_MATRIX\")\nprint(\"=\"*60)\nprint(\"\\nV6 AGGRESSIVE MODE:\")\nprint(\"  - Outputs RLE-encoded masks (not just 'forged')\")\nprint(\"  - Competition-compliant format\")\nprint(\"  - Edge + variance based detection\")\nprint(\"  - 95th percentile threshold\")\nprint(\"\\nWayneIA: The AND is the AGI\")"}], "metadata": {"kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}, "language_info": {"name": "python", "version": "3.10.0"}}, "nbformat": 4, "nbformat_minor": 4}