{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"\n    background: linear-gradient(135deg, #1a1f2c 0%, #2d3748 50%, #4a5568 100%);\n    border: 2px solid #63b3ed;\n    border-radius: 15px;\n    padding: 25px;\n    margin: 20px 0;\n    box-shadow: 0 0 30px rgba(99, 179, 237, 0.4),\n                inset 0 0 20px rgba(255, 255, 255, 0.1);\n    color: #f1f5f9;\n    font-family: 'Segoe UI', system-ui, sans-serif;\n    position: relative;\n    overflow: hidden;\n\">\n\n<div style=\"\n    position: absolute;\n    top: -20px;\n    right: -20px;\n    width: 100px;\n    height: 100px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.25) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<div style=\"\n    position: absolute;\n    bottom: -40px;\n    left: -40px;\n    width: 120px;\n    height: 120px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.2) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<h1 style=\"\n    color: #63b3ed;\n    margin: 0 0 20px 0;\n    text-align: center;\n    font-weight: 700;\n    font-size: 1.8em;\n    text-shadow: 0 0 15px rgba(99, 179, 237, 0.6);\n    position: relative;\n    z-index: 1;\n\">\n    📊 Baseline Strategy: Understanding RLE & Simple Submission\n</h1>\n\n<div style=\"\n    background: rgba(99, 179, 237, 0.1);\n    border-left: 4px solid #63b3ed;\n    border-radius: 8px;\n    padding: 20px;\n    margin: 20px 0;\n    position: relative;\n    z-index: 1;\n\">\n    <h3 style=\"\n        color: #63b3ed;\n        margin-top: 0;\n        font-size: 1.3em;\n        display: flex;\n        align-items: center;\n        gap: 10px;\n    \">\n        🎯 What we'll do in this notebook:\n    </h3>\n    <ul style=\"\n        color: #f1f5f9;\n        font-size: 1.1em;\n        line-height: 1.6;\n        margin-bottom: 0;\n    \">\n        <li>🧪 Understand RLE metric with practical examples</li>\n        <li>🚀 Create a simple \"authentic-only\" submission</li>\n        <li>📈 Learn why this strategy works in some competitions</li>\n        <li>🎲 Test our baseline on the leaderboard(score 0.30 or 30% f1-score)</li>\n    </ul>\n</div>\n\n<div style=\"\n    background: rgba(255, 255, 255, 0.05);\n    border-radius: 10px;\n    padding: 20px;\n    position: relative;\n    z-index: 1;\n\">\n    <h3 style=\"\n        color: #63b3ed;\n        margin-top: 0;\n        font-size: 1.3em;\n        display: flex;\n        align-items: center;\n        gap: 10px;\n    \">\n        💡 Why \"authentic-only\" submission?\n    </h3>\n    <p style=\"color: #f1f5f9; font-size: 1.1em; line-height: 1.6;\">\n        <strong>Experienced competitors often start with this approach!</strong> In many datasets, \n        the majority of images don't contain any objects/forgeries. By submitting \"authentic\" for all images, \n        we get a baseline score that helps us understand the data distribution.\n    </p>\n    \n<div style=\"\n        background: rgba(99, 179, 237, 0.15);\n        border-radius: 8px;\n        padding: 15px;\n        margin: 15px 0;\n    \">\n        <h4 style=\"color: #63b3ed; margin-top: 0;\">When this strategy works well:</h4>\n        <ul style=\"color: #f1f5f9; line-height: 1.5;\">\n            <li>📊 <strong>Imbalanced datasets</strong> - when most images are truly \"authentic\"</li>\n            <li>⚡ <strong>Quick baseline</strong> - to test submission pipeline</li>\n            <li>📈 <strong>Metric understanding</strong> - see how the scoring system works</li>\n            <li>🔍 <strong>Data exploration</strong> - understand the competition dynamics</li>\n        </ul>\n</div>\n    \n<div style=\"\n        background: rgba(247, 127, 127, 0.15);\n        border-radius: 8px;\n        padding: 15px;\n        margin: 15px 0;\n    \">\n        <h4 style=\"color: #f77f7f; margin-top: 0;\">⚠️ Important note:</h4>\n        <p style=\"color: #f1f5f9; margin: 0;\">\n            This is just a <strong>starting point</strong>! While it gives us a quick baseline, \n            to actually compete we'll need to build proper segmentation models. But first, \n            let's make sure our submission pipeline works correctly!\n        </p>\n</div>\n</div>\n</div>","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:14.196869Z","iopub.execute_input":"2025-10-24T00:17:14.197246Z","iopub.status.idle":"2025-10-24T00:17:14.202738Z","shell.execute_reply.started":"2025-10-24T00:17:14.197221Z","shell.execute_reply":"2025-10-24T00:17:14.201648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle_encode(mask, fg_val=1):\n    \"\"\"\n    Convert binary mask to RLE using the competition metric format\n    \"\"\"\n    dots = np.where(mask.T.flatten() == fg_val)[0]\n    run_lengths = []\n    prev = -2\n    \n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    \n    return run_lengths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:14.204588Z","iopub.execute_input":"2025-10-24T00:17:14.204962Z","iopub.status.idle":"2025-10-24T00:17:14.227601Z","shell.execute_reply.started":"2025-10-24T00:17:14.204939Z","shell.execute_reply":"2025-10-24T00:17:14.226196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_mask(mask, title):\n    \"\"\"Visualize mask\"\"\"\n    plt.figure(figsize=(6, 6))\n    plt.imshow(mask, cmap='gray', vmin=0, vmax=1)\n    plt.title(title)\n    plt.axis('off')\n    \n    # Add grid for clarity\n    for i in range(mask.shape[0] + 1):\n        plt.axhline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n        plt.axvline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n    \n    # Show pixel values\n    for i in range(mask.shape[0]):\n        for j in range(mask.shape[1]):\n            plt.text(j, i, str(mask[i, j]), ha='center', va='center', \n                    color='blue' if mask[i, j] == 0 else 'white', fontweight='bold')\n    \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:14.228874Z","iopub.execute_input":"2025-10-24T00:17:14.229465Z","iopub.status.idle":"2025-10-24T00:17:14.262519Z","shell.execute_reply.started":"2025-10-24T00:17:14.229439Z","shell.execute_reply":"2025-10-24T00:17:14.260279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"our_example = np.array([\n    [1, 0],\n    [1, 1]\n])\n\nprint(f'Our example: {our_example}')\nprint(f\"\\nRLE encoding: {rle_encode(our_example)}\")\nvisualize_mask(our_example, \"Our mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:14.26628Z","iopub.execute_input":"2025-10-24T00:17:14.266898Z","iopub.status.idle":"2025-10-24T00:17:14.505587Z","shell.execute_reply.started":"2025-10-24T00:17:14.266868Z","shell.execute_reply":"2025-10-24T00:17:14.503606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create PLUS mask (9x9)\nplus_mask = np.zeros((9, 9), dtype=np.uint8)\n# Vertical line\nplus_mask[2:7, 4] = 1\n# Horizontal line  \nplus_mask[4, 2:7] = 1\n\nprint(plus_mask)\nprint(f\"\\nRLE encoding: {rle_encode(plus_mask)}\")\nvisualize_mask(plus_mask, \"Plus - a mask for segmentation\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:14.506363Z","iopub.execute_input":"2025-10-24T00:17:14.50732Z","iopub.status.idle":"2025-10-24T00:17:14.818709Z","shell.execute_reply.started":"2025-10-24T00:17:14.507284Z","shell.execute_reply":"2025-10-24T00:17:14.817414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create MINUS mask (9x9)\nminus_mask = np.zeros((9, 9), dtype=np.uint8)\n# Horizontal line\nminus_mask[4, 2:7] = 1\n\nprint(minus_mask)\nprint(f\"\\nRLE encoding: {rle_encode(minus_mask)}\")\nvisualize_mask(minus_mask, \"Minus - segmentation mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:14.819629Z","iopub.execute_input":"2025-10-24T00:17:14.819865Z","iopub.status.idle":"2025-10-24T00:17:15.076396Z","shell.execute_reply.started":"2025-10-24T00:17:14.819847Z","shell.execute_reply":"2025-10-24T00:17:15.074919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detailed RLE explanation for plus\nprint(\"Plus mask (9x9):\")\nfor i in range(9):\n    row = ''\n    for j in range(9):\n        row += f\"{plus_mask[i, j]} \"\n    print(row)\n\nprint(f\"\\n1. Flatten to string:\")\nflat_plus = plus_mask.flatten()\nprint(' '.join(map(str, flat_plus)))\n\nprint(f\"\\n2. Split into sequences:\")\n# Add zeros at borders for correct boundary detection\npadded = np.concatenate([[0], flat_plus, [0]])\nchanges = np.where(padded[1:] != padded[:-1])[0] + 1\nruns = changes.copy()\nruns[1::2] -= runs[::2]\n\nprint(f\"Change positions: {changes}\")\nprint(f\"Sequence lengths: {runs}\")\n\nprint(f\"\\n3. Final RLE: '{rle_encode(plus_mask)}'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:15.077666Z","iopub.execute_input":"2025-10-24T00:17:15.078087Z","iopub.status.idle":"2025-10-24T00:17:15.089279Z","shell.execute_reply.started":"2025-10-24T00:17:15.077913Z","shell.execute_reply":"2025-10-24T00:17:15.088012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n    background: linear-gradient(135deg, #1a1f2c 0%, #2d3748 50%, #4a5568 100%);\n    border: 2px solid #63b3ed;\n    border-radius: 15px;\n    padding: 25px;\n    margin: 20px 0;\n    box-shadow: 0 0 30px rgba(99, 179, 237, 0.4),\n                inset 0 0 20px rgba(255, 255, 255, 0.1);\n    color: #f1f5f9;\n    font-family: 'Segoe UI', system-ui, sans-serif;\n    position: relative;\n    overflow: hidden;\n\">\n\n<div style=\"\n    position: absolute;\n    top: -20px;\n    right: -20px;\n    width: 100px;\n    height: 100px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.25) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<div style=\"\n    position: absolute;\n    bottom: -40px;\n    left: -40px;\n    width: 120px;\n    height: 120px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.2) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<h1 style=\"\n    color: #63b3ed;\n    margin: 0 0 20px 0;\n    text-align: center;\n    font-weight: 700;\n    font-size: 1.8em;\n    text-shadow: 0 0 15px rgba(99, 179, 237, 0.6);\n    position: relative;\n    z-index: 1;\n\">\n    Create sumission distributed by the most frequent position in the mask\n</h1>","metadata":{}},{"cell_type":"code","source":"np.random.seed(73)\n\ndef mask_distribution():\n    all_norm_positions = []\n    heatmap_size = (100, 100)\n    heatmap = np.zeros(heatmap_size, dtype=np.float32)\n    \n    train_masks_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_masks'\n    \n    if not os.path.exists(train_masks_dir):\n        return (0.5, 0.5), None\n    \n    for mask_file in os.listdir(train_masks_dir):\n        if mask_file.endswith('.npy'):\n            mask_path = os.path.join(train_masks_dir, mask_file)\n            try:\n                mask = np.load(mask_path)\n                \n                if mask.ndim == 3:\n                    if mask.shape[0] == 1:\n                        mask = mask[0]\n                    elif mask.shape[2] == 1:\n                        mask = mask[:, :, 0]\n                    else:\n                        mask = (mask == 1).astype(np.uint8)\n                        if mask.ndim == 3:\n                            mask = mask[:, :, 0] if mask.shape[2] == 1 else mask[:, :, 0]\n                \n                if mask.ndim != 2:\n                    continue\n                \n                y_coords, x_coords = np.where(mask > 0)\n                \n                if len(y_coords) > 0:\n                    height, width = mask.shape\n                    \n                    for y, x in zip(y_coords, x_coords):\n                        norm_y = y / height\n                        norm_x = x / width\n                        \n                        heatmap_y = int(norm_y * heatmap_size[0])\n                        heatmap_x = int(norm_x * heatmap_size[1])\n                        \n                        heatmap_y = min(heatmap_y, heatmap_size[0] - 1)\n                        heatmap_x = min(heatmap_x, heatmap_size[1] - 1)\n                        \n                        heatmap[heatmap_y, heatmap_x] += 1\n                        all_norm_positions.append((norm_x, norm_y))\n                        \n            except Exception as e:\n                continue\n    \n    if all_norm_positions:\n        max_heatmap_pos = np.unravel_index(np.argmax(heatmap), heatmap.shape)\n        max_norm_y = max_heatmap_pos[0] / heatmap_size[0]\n        max_norm_x = max_heatmap_pos[1] / heatmap_size[1]\n        return (max_norm_x, max_norm_y), heatmap\n    \n    return (0.5, 0.5), heatmap\n\nhottest_norm_pos, heatmap = mask_distribution()\n\nif heatmap is not None:\n    plt.figure(figsize=(10, 8))\n    plt.imshow(heatmap, cmap='hot', interpolation='nearest')\n    plt.colorbar()\n    plt.title('Forgery Location Heatmap')\n    plt.xlabel('Normalized X')\n    plt.ylabel('Normalized Y')\n    plt.show()\n\ntest_images_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images'\nsample_submission = pd.read_csv('/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T00:17:15.091557Z","iopub.execute_input":"2025-10-24T00:17:15.091887Z","iopub.status.idle":"2025-10-24T00:17:15.132384Z","shell.execute_reply.started":"2025-10-24T00:17:15.091867Z","shell.execute_reply":"2025-10-24T00:17:15.13126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_data = []\nfor case_id in sample_submission['case_id']:\n    img_path = os.path.join(test_images_dir, f\"{case_id}.png\")\n    \n    with Image.open(img_path) as img:\n        width, height = img.size\n    \n    if np.random.random() < 0.01:\n        mask = np.zeros((height, width), dtype=np.uint8)\n        \n        offset_x = np.random.uniform(-0.3, 0.3) * width\n        offset_y = np.random.uniform(-0.3, 0.3) * height\n        \n        center_x = int(hottest_norm_pos[0] * width + offset_x)\n        center_y = int(hottest_norm_pos[1] * height + offset_y)\n        \n        h = 4\n        w = 4\n        \n        y0 = max(0, center_y - h//2)\n        x0 = max(0, center_x - w//2)\n        y1 = min(height, y0 + h)\n        x1 = min(width, x0 + w)\n        \n        actual_h = y1 - y0\n        actual_w = x1 - x0\n        \n        if actual_h > 0 and actual_w > 0:\n            mask[y0:y1, x0:x1] = 1\n        \n        RLE_res = rle_encode(mask)\n        res = [int(x) for x in RLE_res]\n        annotation = json.dumps(res)\n    else:\n        annotation = 'authentic'\n    \n    submission_data.append({\n        'case_id': case_id,\n        'annotation': annotation\n    })\n\nsubmission = pd.DataFrame(submission_data)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n    background: linear-gradient(135deg, #1a1f2c 0%, #2d3748 50%, #4a5568 100%);\n    border: 2px solid #63b3ed;\n    border-radius: 15px;\n    padding: 25px;\n    margin: 20px 0;\n    box-shadow: 0 0 30px rgba(99, 179, 237, 0.4),\n                inset 0 0 20px rgba(255, 255, 255, 0.1);\n    color: #f1f5f9;\n    font-family: 'Segoe UI', system-ui, sans-serif;\n    position: relative;\n    overflow: hidden;\n\">\n\n<div style=\"\n    position: absolute;\n    top: -20px;\n    right: -20px;\n    width: 100px;\n    height: 100px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.25) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<div style=\"\n    position: absolute;\n    bottom: -40px;\n    left: -40px;\n    width: 120px;\n    height: 120px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.2) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<h1 style=\"\n    color: #63b3ed;\n    margin: 0 0 20px 0;\n    text-align: center;\n    font-weight: 700;\n    font-size: 1.8em;\n    text-shadow: 0 0 15px rgba(99, 179, 237, 0.6);\n    position: relative;\n    z-index: 1;\n\">\n    If i have mistake write pls in comments\n</h1>","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}