{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:43.251354Z","iopub.execute_input":"2025-11-01T02:11:43.251733Z","iopub.status.idle":"2025-11-01T02:11:43.636483Z","shell.execute_reply.started":"2025-11-01T02:11:43.251700Z","shell.execute_reply":"2025-11-01T02:11:43.635430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle_encode(mask, fg_val=1):\n    \"\"\"\n    Convert binary mask to RLE using the competition metric format\n    \"\"\"\n    dots = np.where(mask.T.flatten() == fg_val)[0]\n    run_lengths = []\n    prev = -2\n    \n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    \n    return run_lengths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:45.596787Z","iopub.execute_input":"2025-11-01T02:11:45.597382Z","iopub.status.idle":"2025-11-01T02:11:45.604443Z","shell.execute_reply.started":"2025-11-01T02:11:45.597345Z","shell.execute_reply":"2025-11-01T02:11:45.603561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_mask(mask, title):\n    \"\"\"Visualize mask\"\"\"\n    plt.figure(figsize=(6, 6))\n    plt.imshow(mask, cmap='gray', vmin=0, vmax=1)\n    plt.title(title)\n    plt.axis('off')\n    \n    # Add grid for clarity\n    for i in range(mask.shape[0] + 1):\n        plt.axhline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n        plt.axvline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n    \n    # Show pixel values\n    for i in range(mask.shape[0]):\n        for j in range(mask.shape[1]):\n            plt.text(j, i, str(mask[i, j]), ha='center', va='center', \n                    color='blue' if mask[i, j] == 0 else 'white', fontweight='bold')\n    \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:46.686592Z","iopub.execute_input":"2025-11-01T02:11:46.687202Z","iopub.status.idle":"2025-11-01T02:11:46.698054Z","shell.execute_reply.started":"2025-11-01T02:11:46.687141Z","shell.execute_reply":"2025-11-01T02:11:46.696063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"our_example = np.array([\n    [1, 0],\n    [1, 1]\n])\n\nprint(f'Our example: {our_example}')\nprint(f\"\\nRLE encoding: {rle_encode(our_example)}\")\nvisualize_mask(our_example, \"Our mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:47.052254Z","iopub.execute_input":"2025-11-01T02:11:47.052619Z","iopub.status.idle":"2025-11-01T02:11:47.311998Z","shell.execute_reply.started":"2025-11-01T02:11:47.052593Z","shell.execute_reply":"2025-11-01T02:11:47.310929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create PLUS mask (9x9)\nplus_mask = np.zeros((9, 9), dtype=np.uint8)\n# Vertical line\nplus_mask[2:7, 4] = 1\n# Horizontal line  \nplus_mask[4, 2:7] = 1\n\nprint(plus_mask)\nprint(f\"\\nRLE encoding: {rle_encode(plus_mask)}\")\nvisualize_mask(plus_mask, \"Plus - a mask for segmentation\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:47.435937Z","iopub.execute_input":"2025-11-01T02:11:47.436263Z","iopub.status.idle":"2025-11-01T02:11:47.709438Z","shell.execute_reply.started":"2025-11-01T02:11:47.436242Z","shell.execute_reply":"2025-11-01T02:11:47.707841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create MINUS mask (9x9)\nminus_mask = np.zeros((9, 9), dtype=np.uint8)\n# Horizontal line\nminus_mask[4, 2:7] = 1\n\nprint(minus_mask)\nprint(f\"\\nRLE encoding: {rle_encode(minus_mask)}\")\nvisualize_mask(minus_mask, \"Minus - segmentation mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:47.792599Z","iopub.execute_input":"2025-11-01T02:11:47.792931Z","iopub.status.idle":"2025-11-01T02:11:48.121369Z","shell.execute_reply.started":"2025-11-01T02:11:47.792907Z","shell.execute_reply":"2025-11-01T02:11:48.120420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detailed RLE explanation for plus\nprint(\"Plus mask (9x9):\")\nfor i in range(9):\n    row = ''\n    for j in range(9):\n        row += f\"{plus_mask[i, j]} \"\n    print(row)\n\nprint(f\"\\n1. Flatten to string:\")\nflat_plus = plus_mask.flatten()\nprint(' '.join(map(str, flat_plus)))\n\nprint(f\"\\n2. Split into sequences:\")\n# Add zeros at borders for correct boundary detection\npadded = np.concatenate([[0], flat_plus, [0]])\nchanges = np.where(padded[1:] != padded[:-1])[0] + 1\nruns = changes.copy()\nruns[1::2] -= runs[::2]\n\nprint(f\"Change positions: {changes}\")\nprint(f\"Sequence lengths: {runs}\")\n\nprint(f\"\\n3. Final RLE: '{rle_encode(plus_mask)}'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:11:51.247716Z","iopub.execute_input":"2025-11-01T02:11:51.248081Z","iopub.status.idle":"2025-11-01T02:11:51.257743Z","shell.execute_reply.started":"2025-11-01T02:11:51.248054Z","shell.execute_reply":"2025-11-01T02:11:51.256485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n    background: linear-gradient(135deg, #1a1f2c 0%, #2d3748 50%, #4a5568 100%);\n    border: 2px solid #63b3ed;\n    border-radius: 15px;\n    padding: 25px;\n    margin: 20px 0;\n    box-shadow: 0 0 30px rgba(99, 179, 237, 0.4),\n                inset 0 0 20px rgba(255, 255, 255, 0.1);\n    color: #f1f5f9;\n    font-family: 'Segoe UI', system-ui, sans-serif;\n    position: relative;\n    overflow: hidden;\n\">\n\n<div style=\"\n    position: absolute;\n    top: -20px;\n    right: -20px;\n    width: 100px;\n    height: 100px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.25) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<div style=\"\n    position: absolute;\n    bottom: -40px;\n    left: -40px;\n    width: 120px;\n    height: 120px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.2) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<h1 style=\"\n    color: #63b3ed;\n    margin: 0 0 20px 0;\n    text-align: center;\n    font-weight: 700;\n    font-size: 1.8em;\n    text-shadow: 0 0 15px rgba(99, 179, 237, 0.6);\n    position: relative;\n    z-index: 1;\n\">\n    Create sumission distributed by the most frequent position in the mask\n</h1>","metadata":{}},{"cell_type":"code","source":"np.random.seed(71)\n\ndef mask_distribution():\n    all_norm_positions = []\n    heatmap_size = (100, 100)\n    heatmap = np.zeros(heatmap_size, dtype=np.float32)\n    \n    train_masks_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_masks'\n    \n    if not os.path.exists(train_masks_dir):\n        return (0.5, 0.5), None\n    \n    for mask_file in os.listdir(train_masks_dir):\n        if mask_file.endswith('.npy'):\n            mask_path = os.path.join(train_masks_dir, mask_file)\n            try:\n                mask = np.load(mask_path)\n                \n                if mask.ndim == 3:\n                    if mask.shape[0] == 1:\n                        mask = mask[0]\n                    elif mask.shape[2] == 1:\n                        mask = mask[:, :, 0]\n                    else:\n                        mask = (mask == 1).astype(np.uint8)\n                        if mask.ndim == 3:\n                            mask = mask[:, :, 0] if mask.shape[2] == 1 else mask[:, :, 0]\n                \n                if mask.ndim != 2:\n                    continue\n                \n                y_coords, x_coords = np.where(mask > 0)\n                \n                if len(y_coords) > 0:\n                    height, width = mask.shape\n                    \n                    for y, x in zip(y_coords, x_coords):\n                        norm_y = y / height\n                        norm_x = x / width\n                        \n                        heatmap_y = int(norm_y * heatmap_size[0])\n                        heatmap_x = int(norm_x * heatmap_size[1])\n                        \n                        heatmap_y = min(heatmap_y, heatmap_size[0] - 1)\n                        heatmap_x = min(heatmap_x, heatmap_size[1] - 1)\n                        \n                        heatmap[heatmap_y, heatmap_x] += 1\n                        all_norm_positions.append((norm_x, norm_y))\n                        \n            except Exception as e:\n                continue\n    \n    if all_norm_positions:\n        max_heatmap_pos = np.unravel_index(np.argmax(heatmap), heatmap.shape)\n        max_norm_y = max_heatmap_pos[0] / heatmap_size[0]\n        max_norm_x = max_heatmap_pos[1] / heatmap_size[1]\n        return (max_norm_x, max_norm_y), heatmap\n    \n    return (0.5, 0.5), heatmap\n\nhottest_norm_pos, heatmap = mask_distribution()\n\nif heatmap is not None:\n    plt.figure(figsize=(10, 8))\n    plt.imshow(heatmap, cmap='hot', interpolation='nearest')\n    plt.colorbar()\n    plt.title('Forgery Location Heatmap')\n    plt.xlabel('Normalized X')\n    plt.ylabel('Normalized Y')\n    plt.show()\n\ntest_images_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images'\nsample_submission = pd.read_csv('/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:12:06.327883Z","iopub.execute_input":"2025-11-01T02:12:06.328233Z","iopub.status.idle":"2025-11-01T02:18:44.706337Z","shell.execute_reply.started":"2025-11-01T02:12:06.328202Z","shell.execute_reply":"2025-11-01T02:18:44.705167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_data = []\nfor case_id in sample_submission['case_id']:\n    img_path = os.path.join(test_images_dir, f\"{case_id}.png\")\n    \n    with Image.open(img_path) as img:\n        width, height = img.size\n    \n    if np.random.random() < 0.01:\n        mask = np.zeros((height, width), dtype=np.uint8)\n        \n        offset_x = np.random.uniform(-0.3, 0.3) * width\n        offset_y = np.random.uniform(-0.3, 0.3) * height\n        \n        center_x = int(hottest_norm_pos[0] * width + offset_x)\n        center_y = int(hottest_norm_pos[1] * height + offset_y)\n        \n        h = 4\n        w = 4\n        \n        y0 = max(0, center_y - h//2)\n        x0 = max(0, center_x - w//2)\n        y1 = min(height, y0 + h)\n        x1 = min(width, x0 + w)\n        \n        actual_h = y1 - y0\n        actual_w = x1 - x0\n        \n        if actual_h > 0 and actual_w > 0:\n            mask[y0:y1, x0:x1] = 1\n        \n        RLE_res = rle_encode(mask)\n        res = [int(x) for x in RLE_res]\n        annotation = json.dumps(res)\n    else:\n        annotation = 'authentic'\n    \n    submission_data.append({\n        'case_id': case_id,\n        'annotation': annotation\n    })\n\nsubmission = pd.DataFrame(submission_data)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:18:44.777655Z","iopub.execute_input":"2025-11-01T02:18:44.778024Z","iopub.status.idle":"2025-11-01T02:18:44.791840Z","shell.execute_reply.started":"2025-11-01T02:18:44.777994Z","shell.execute_reply":"2025-11-01T02:18:44.790869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\n# Example setup\n# (Make sure you have these defined)\n# sample_submission = pd.read_csv(\"sample_submission.csv\")\n# test_images_dir = \"path/to/test/images\"\n# hottest_norm_pos = (0.5, 0.5)  # Example center point (normalized 0–1)\n\ndef rle_encode(mask):\n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return runs\n\nsubmission_data = []\n\nfor case_id in sample_submission['case_id']:\n    img_path = os.path.join(test_images_dir, f\"{case_id}.png\")\n\n    with Image.open(img_path) as img:\n        width, height = img.size\n        img_array = np.array(img)\n\n    if np.random.random() < 0.01:\n        mask = np.zeros((height, width), dtype=np.uint8)\n        offset_x = np.random.uniform(-0.3, 0.3) * width\n        offset_y = np.random.uniform(-0.3, 0.3) * height\n        center_x = int(hottest_norm_pos[0] * width + offset_x)\n        center_y = int(hottest_norm_pos[1] * height + offset_y)\n        h = 4\n        w = 4\n\n        y0 = max(0, center_y - h//2)\n        x0 = max(0, center_x - w//2)\n        y1 = min(height, y0 + h)\n        x1 = min(width, x0 + w)\n        actual_h = y1 - y0\n        actual_w = x1 - x0\n\n        if actual_h > 0 and actual_w > 0:\n            mask[y0:y1, x0:x1] = 1\n            RLE_res = rle_encode(mask)\n            res = [int(x) for x in RLE_res]\n            annotation = json.dumps(res)\n\n            # 🔹 PLOT: visualize the image and mask overlay\n            plt.figure(figsize=(5, 5))\n            plt.imshow(img_array, cmap='gray')\n            plt.imshow(mask, alpha=0.4, cmap='Reds')  # overlay in red\n            plt.title(f\"Mask overlay for {case_id}\")\n            plt.axis('off')\n            plt.show()\n\n        else:\n            annotation = 'authentic'\n    else:\n        annotation = 'authentic'\n\n    submission_data.append({\n        'case_id': case_id,\n        'annotation': annotation\n    })\n\nsubmission = pd.DataFrame(submission_data)\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"✅ Submission file saved as 'submission.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:18:44.708343Z","iopub.execute_input":"2025-11-01T02:18:44.708680Z","iopub.status.idle":"2025-11-01T02:18:44.776660Z","shell.execute_reply.started":"2025-11-01T02:18:44.708659Z","shell.execute_reply":"2025-11-01T02:18:44.775432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport matplotlib.patches as patches\n\n# --------------------------\n# SETUP: define your paths\n# --------------------------\n# sample_submission = pd.read_csv(\"sample_submission.csv\")  # your submission template\n# test_images_dir = \"path/to/test/images\"                  # folder with test images\n# hottest_norm_pos = (0.5, 0.5)                            # normalized center (example)\n\n# Make folder to save plots\nos.makedirs(\"plots\", exist_ok=True)\n\n# --------------------------\n# RLE Encode function\n# --------------------------\ndef rle_encode(mask):\n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return runs\n\n# --------------------------\n# MAIN LOOP\n# --------------------------\nsubmission_data = []\n\nfor case_id in sample_submission['case_id']:\n    img_path = os.path.join(test_images_dir, f\"{case_id}.png\")\n\n    with Image.open(img_path) as img:\n        width, height = img.size\n        img_array = np.array(img)\n\n    if np.random.random() < 0.01:\n        mask = np.zeros((height, width), dtype=np.uint8)\n        offset_x = np.random.uniform(-0.3, 0.3) * width\n        offset_y = np.random.uniform(-0.3, 0.3) * height\n        center_x = int(hottest_norm_pos[0] * width + offset_x)\n        center_y = int(hottest_norm_pos[1] * height + offset_y)\n        h = 4\n        w = 4\n\n        y0 = max(0, center_y - h//2)\n        x0 = max(0, center_x - w//2)\n        y1 = min(height, y0 + h)\n        x1 = min(width, x0 + w)\n        actual_h = y1 - y0\n        actual_w = x1 - x0\n\n        if actual_h > 0 and actual_w > 0:\n            mask[y0:y1, x0:x1] = 1\n            RLE_res = rle_encode(mask)\n            res = [int(x) for x in RLE_res]\n            annotation = json.dumps(res)\n\n            # --------------------------\n            # PLOT: Image + Mask + Bounding Box\n            # --------------------------\n            fig, ax = plt.subplots(figsize=(6, 6))\n            ax.imshow(img_array, cmap='gray')\n            ax.imshow(mask, alpha=0.4, cmap='Reds')\n            rect = patches.Rectangle((x0, y0), actual_w, actual_h,\n                                     linewidth=2, edgecolor='cyan', facecolor='none')\n            ax.add_patch(rect)\n            ax.set_title(f\"Mask Overlay for {case_id}\")\n            ax.axis('off')\n\n            # Save plot\n            plt.savefig(f\"plots/{case_id}_overlay.png\", bbox_inches='tight')\n            plt.close()\n\n        else:\n            annotation = 'authentic'\n    else:\n        annotation = 'authentic'\n\n    submission_data.append({\n        'case_id': case_id,\n        'annotation': annotation\n    })\n\n# --------------------------\n# SAVE FINAL SUBMISSION\n# --------------------------\nsubmission = pd.DataFrame(submission_data)\nsubmission.to_csv('submission.csv', index=False)\nprint(\"✅ Submission file saved as 'submission.csv'\")\nprint(\"✅ Mask overlay plots saved in 'plots/' folder\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-01T02:18:44.793833Z","iopub.execute_input":"2025-11-01T02:18:44.794137Z","iopub.status.idle":"2025-11-01T02:18:44.852600Z","shell.execute_reply.started":"2025-11-01T02:18:44.794114Z","shell.execute_reply":"2025-11-01T02:18:44.851167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}