{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:13.593826Z","iopub.execute_input":"2025-10-30T03:39:13.594050Z","iopub.status.idle":"2025-10-30T03:39:13.946514Z","shell.execute_reply.started":"2025-10-30T03:39:13.594027Z","shell.execute_reply":"2025-10-30T03:39:13.945533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef rle_encode(mask, fg_val=1):\n    \"\"\"\n    Convert a binary mask to RLE (Run-Length Encoding) using a competition-style format.\n\n    Args:\n        mask (np.ndarray): 2D binary mask where foreground pixels are fg_val.\n        fg_val (int, optional): Value considered as foreground. Default is 1.\n\n    Returns:\n        list: RLE as a list of (start_position, run_length) pairs.\n    \"\"\"\n    # Flatten the mask in column-major order (Fortran-style) and find foreground indices\n    dots = np.where(mask.T.flatten() == fg_val)[0]\n\n    run_lengths = []\n    prev = -2\n\n    for b in dots:\n        if b > prev + 1:\n            # Start a new run\n            run_lengths.extend((b + 1, 0))  # RLE positions are 1-indexed\n        run_lengths[-1] += 1  # Increase the length of the current run\n        prev = b\n\n    return run_lengths\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:41.619668Z","iopub.execute_input":"2025-10-30T03:39:41.619958Z","iopub.status.idle":"2025-10-30T03:39:41.626094Z","shell.execute_reply.started":"2025-10-30T03:39:41.619936Z","shell.execute_reply":"2025-10-30T03:39:41.625256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef visualize_mask(mask, title=\"Mask\"):\n    \"\"\"\n    Visualize a binary mask with grid and pixel values.\n\n    Args:\n        mask (np.ndarray): 2D array representing the mask.\n        title (str): Title for the plot.\n    \"\"\"\n    plt.figure(figsize=(6, 6))\n    plt.imshow(mask, cmap='gray', vmin=0, vmax=1)\n    plt.title(title)\n    plt.axis('off')\n\n    # Add grid for clarity\n    for i in range(mask.shape[0] + 1):\n        plt.axhline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n        plt.axvline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n\n    # Show pixel values\n    for i in range(mask.shape[0]):\n        for j in range(mask.shape[1]):\n            plt.text(\n                j, i, str(mask[i, j]),\n                ha='center', va='center',\n                color='blue' if mask[i, j] == 0 else 'white',\n                fontweight='bold'\n            )\n\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:44.142046Z","iopub.execute_input":"2025-10-30T03:39:44.142769Z","iopub.status.idle":"2025-10-30T03:39:44.148935Z","shell.execute_reply.started":"2025-10-30T03:39:44.142747Z","shell.execute_reply":"2025-10-30T03:39:44.148152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"our_example = np.array([\n    [1, 0],\n    [1, 1]\n])\n\nprint(f'Our example: {our_example}')\nprint(f\"\\nRLE encoding: {rle_encode(our_example)}\")\nvisualize_mask(our_example, \"Our mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:47.361556Z","iopub.execute_input":"2025-10-30T03:39:47.362376Z","iopub.status.idle":"2025-10-30T03:39:47.569834Z","shell.execute_reply.started":"2025-10-30T03:39:47.362351Z","shell.execute_reply":"2025-10-30T03:39:47.568821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create PLUS mask (9x9)\nplus_mask = np.zeros((9, 9), dtype=np.uint8)\n# Vertical line\nplus_mask[2:7, 4] = 1\n# Horizontal line  \nplus_mask[4, 2:7] = 1\n\nprint(plus_mask)\nprint(f\"\\nRLE encoding: {rle_encode(plus_mask)}\")\nvisualize_mask(plus_mask, \"Plus - a mask for segmentation\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:50.463073Z","iopub.execute_input":"2025-10-30T03:39:50.463342Z","iopub.status.idle":"2025-10-30T03:39:50.695477Z","shell.execute_reply.started":"2025-10-30T03:39:50.463322Z","shell.execute_reply":"2025-10-30T03:39:50.694484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create MINUS mask (9x9)\nminus_mask = np.zeros((9, 9), dtype=np.uint8)\n# Horizontal line\nminus_mask[4, 2:7] = 1\n\nprint(minus_mask)\nprint(f\"\\nRLE encoding: {rle_encode(minus_mask)}\")\nvisualize_mask(minus_mask, \"Minus - segmentation mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:53.267148Z","iopub.execute_input":"2025-10-30T03:39:53.267420Z","iopub.status.idle":"2025-10-30T03:39:53.554576Z","shell.execute_reply.started":"2025-10-30T03:39:53.267401Z","shell.execute_reply":"2025-10-30T03:39:53.553754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detailed RLE explanation for plus\nprint(\"Plus mask (9x9):\")\nfor i in range(9):\n    row = ''\n    for j in range(9):\n        row += f\"{plus_mask[i, j]} \"\n    print(row)\n\nprint(f\"\\n1. Flatten to string:\")\nflat_plus = plus_mask.flatten()\nprint(' '.join(map(str, flat_plus)))\n\nprint(f\"\\n2. Split into sequences:\")\n# Add zeros at borders for correct boundary detection\npadded = np.concatenate([[0], flat_plus, [0]])\nchanges = np.where(padded[1:] != padded[:-1])[0] + 1\nruns = changes.copy()\nruns[1::2] -= runs[::2]\n\nprint(f\"Change positions: {changes}\")\nprint(f\"Sequence lengths: {runs}\")\n\nprint(f\"\\n3. Final RLE: '{rle_encode(plus_mask)}'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:39:55.577279Z","iopub.execute_input":"2025-10-30T03:39:55.577869Z","iopub.status.idle":"2025-10-30T03:39:55.585525Z","shell.execute_reply.started":"2025-10-30T03:39:55.577845Z","shell.execute_reply":"2025-10-30T03:39:55.584698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Example: Plus-shaped mask (9x9)\nplus_mask = np.array([\n    [0,0,1,0,0,1,0,0,0],\n    [0,0,1,0,0,1,0,0,0],\n    [1,1,1,1,1,1,1,1,1],\n    [0,0,1,0,0,1,0,0,0],\n    [0,0,1,0,0,1,0,0,0],\n    [0,0,1,0,0,1,0,0,0],\n    [0,0,1,0,0,1,0,0,0],\n    [0,0,1,0,0,1,0,0,0],\n    [0,0,1,0,0,1,0,0,0]\n])\n\nprint(\"Step 0: Plus mask (9x9):\")\nfor row in plus_mask:\n    print(' '.join(map(str, row)))\n\n# Step 1: Flatten mask\nflat_plus = plus_mask.flatten()\nprint(\"\\nStep 1: Flattened mask (row-major order):\")\nprint(' '.join(map(str, flat_plus)))\n\n# Step 2: Detect sequences for RLE\n# Add zeros at boundaries to detect changes\npadded = np.concatenate([[0], flat_plus, [0]])\nchanges = np.where(padded[1:] != padded[:-1])[0] + 1  # positions where value changes\nruns = changes.copy()\nruns[1::2] -= runs[::2]  # compute run lengths\n\nprint(\"\\nStep 2: Sequence detection\")\nprint(f\"Change positions: {changes}\")\nprint(f\"Run lengths: {runs}\")\n\n# Step 3: Use your rle_encode function\ndef rle_encode(mask, fg_val=1):\n    \"\"\"\n    Convert binary mask to RLE (1-indexed).\n    \"\"\"\n    dots = np.where(mask.T.flatten() == fg_val)[0]\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\nprint(\"\\nStep 3: Final RLE (competition format):\")\nprint(rle_encode(plus_mask))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:41:50.137646Z","iopub.execute_input":"2025-10-30T03:41:50.137939Z","iopub.status.idle":"2025-10-30T03:41:50.149407Z","shell.execute_reply.started":"2025-10-30T03:41:50.137918Z","shell.execute_reply":"2025-10-30T03:41:50.148378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(61)\n\ndef mask_distribution():\n    all_norm_positions = []\n    heatmap_size = (100, 100)\n    heatmap = np.zeros(heatmap_size, dtype=np.float32)\n    \n    train_masks_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_masks'\n    \n    if not os.path.exists(train_masks_dir):\n        return (0.5, 0.5), None\n    \n    for mask_file in os.listdir(train_masks_dir):\n        if mask_file.endswith('.npy'):\n            mask_path = os.path.join(train_masks_dir, mask_file)\n            try:\n                mask = np.load(mask_path)\n                \n                if mask.ndim == 3:\n                    if mask.shape[0] == 1:\n                        mask = mask[0]\n                    elif mask.shape[2] == 1:\n                        mask = mask[:, :, 0]\n                    else:\n                        mask = (mask == 1).astype(np.uint8)\n                        if mask.ndim == 3:\n                            mask = mask[:, :, 0] if mask.shape[2] == 1 else mask[:, :, 0]\n                \n                if mask.ndim != 2:\n                    continue\n                \n                y_coords, x_coords = np.where(mask > 0)\n                \n                if len(y_coords) > 0:\n                    height, width = mask.shape\n                    \n                    for y, x in zip(y_coords, x_coords):\n                        norm_y = y / height\n                        norm_x = x / width\n                        \n                        heatmap_y = int(norm_y * heatmap_size[0])\n                        heatmap_x = int(norm_x * heatmap_size[1])\n                        \n                        heatmap_y = min(heatmap_y, heatmap_size[0] - 1)\n                        heatmap_x = min(heatmap_x, heatmap_size[1] - 1)\n                        \n                        heatmap[heatmap_y, heatmap_x] += 1\n                        all_norm_positions.append((norm_x, norm_y))\n                        \n            except Exception as e:\n                continue\n    \n    if all_norm_positions:\n        max_heatmap_pos = np.unravel_index(np.argmax(heatmap), heatmap.shape)\n        max_norm_y = max_heatmap_pos[0] / heatmap_size[0]\n        max_norm_x = max_heatmap_pos[1] / heatmap_size[1]\n        return (max_norm_x, max_norm_y), heatmap\n    \n    return (0.5, 0.5), heatmap\n\nhottest_norm_pos, heatmap = mask_distribution()\n\nif heatmap is not None:\n    plt.figure(figsize=(10, 8))\n    plt.imshow(heatmap, cmap='hot', interpolation='nearest')\n    plt.colorbar()\n    plt.title('Forgery Location Heatmap')\n    plt.xlabel('Normalized X')\n    plt.ylabel('Normalized Y')\n    plt.show()\n\ntest_images_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images'\nsample_submission = pd.read_csv('/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv')\n\nsubmission_data = []\nfor case_id in sample_submission['case_id']:\n    img_path = os.path.join(test_images_dir, f\"{case_id}.png\")\n    \n    with Image.open(img_path) as img:\n        width, height = img.size\n    \n    if np.random.random() < 0.01:\n        mask = np.zeros((height, width), dtype=np.uint8)\n        \n        center_x = int(hottest_norm_pos[0] * width)\n        center_y = int(hottest_norm_pos[1] * height)\n        \n        center_x = max(4, min(center_x, width - 5))\n        center_y = max(4, min(center_y, height - 5))\n        \n        max_mask_size = min(width, height) // 20\n        h = min(8, max_mask_size)\n        w = min(8, max_mask_size)\n        \n        y0 = center_y - h//2\n        x0 = center_x - w//2\n        \n        mask[y0:y0+h, x0:x0+w] = 1\n        \n        RLE_res = rle_encode(mask)\n        res = [int(x) for x in RLE_res]\n        annotation = json.dumps(res)\n    else:\n        annotation = 'authentic'\n    \n    submission_data.append({\n        'case_id': case_id,\n        'annotation': annotation\n    })\n\nsubmission = pd.DataFrame(submission_data)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T03:41:54.596196Z","iopub.execute_input":"2025-10-30T03:41:54.596518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}