{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\nimport numpy as np\nimport pandas as pd\nimport os\nfrom pathlib import Path\nimport json\nimport cv2\nfrom tqdm import tqdm\nfrom scipy.fftpack import dct\nfrom scipy.ndimage import median_filter\nfrom skimage.measure import label, regionprops\nfrom skimage.feature import local_binary_pattern, hog\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# GPU-accelerated ML\nimport xgboost as xgb\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nimport joblib\n\n# Check GPU availability\nimport torch\nprint(f\"GPU Available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n\nprint(\"=\"*80)\nprint(\"GPU-ACCELERATED TRADITIONAL ML FORGERY DETECTION\")\nprint(\"=\"*80)\n\n# ============================================================================\n# CONFIGURATION\n# ============================================================================\n\nclass Config:\n    BASE_PATH = Path('/kaggle/input/recodai-luc-scientific-image-forgery-detection')\n    TRAIN_IMAGES_DIR = BASE_PATH / 'train_images'\n    TRAIN_MASKS_DIR = BASE_PATH / 'train_masks'\n    TEST_IMAGES_DIR = BASE_PATH / 'test_images'\n    SAMPLE_SUB_PATH = BASE_PATH / 'sample_submission.csv'\n    \n    # Feature extraction\n    PATCH_SIZE = 64  # Extract features from 64x64 patches\n    PATCHES_PER_IMAGE = 80  # Sample patches per image\n    \n    # Training\n    USE_GPU = True\n    N_ESTIMATORS = 250\n    MAX_DEPTH = 10\n    LEARNING_RATE = 0.1\n    \n    # For segmentation\n    STRIDE = 32  # Sliding window stride for prediction\n    FORGERY_THRESHOLD = 0.5\n    MIN_REGION_AREA = 150\n    \n    # Sampling\n    MAX_TRAIN_SAMPLES = 40000  # Max patches for training\n    MAX_TRAIN_IMAGES = 400  # Limit training images for speed\n    \n    VISUALIZE_SAMPLES = True\n    MAX_VIZ_SAMPLES = 2\n\n# ============================================================================\n# DATA DISCOVERY\n# ============================================================================\n\ndef discover_data(config):\n    \"\"\"Discover all data files\"\"\"\n    print(\"\\n\" + \"=\"*80)\n    print(\"DATA DISCOVERY\")\n    print(\"=\"*80)\n    \n    authentic_dir = config.TRAIN_IMAGES_DIR / 'authentic'\n    forged_dir = config.TRAIN_IMAGES_DIR / 'forged'\n    \n    authentic_images = sorted(list(authentic_dir.glob('*.png'))) if authentic_dir.exists() else []\n    forged_images = sorted(list(forged_dir.glob('*.png'))) if forged_dir.exists() else []\n    \n    print(f\"📁 Training - Authentic: {len(authentic_images)}, Forged: {len(forged_images)}\")\n    \n    # Load masks\n    mask_files = {}\n    if config.TRAIN_MASKS_DIR.exists():\n        for mask_path in config.TRAIN_MASKS_DIR.glob('*.npy'):\n            mask_files[mask_path.stem] = mask_path\n    \n    print(f\"📁 Masks: {len(mask_files)}\")\n    \n    test_images = sorted(list(config.TEST_IMAGES_DIR.glob('*.png')))\n    print(f\"📁 Test: {len(test_images)}\")\n    \n    print(\"=\"*80)\n    \n    return authentic_images, forged_images, mask_files, test_images\n\n# ============================================================================\n# RICH FEATURE EXTRACTION\n# ============================================================================\n\ndef extract_patch_features(patch):\n    \"\"\"Extract comprehensive features from a patch\"\"\"\n    \n    # Validate patch\n    if patch.shape[0] < 8 or patch.shape[1] < 8:\n        return None\n    \n    features = []\n    \n    # Convert to grayscale\n    if patch.ndim == 3:\n        gray = cv2.cvtColor(patch, cv2.COLOR_RGB2GRAY)\n    else:\n        gray = patch\n    \n    gray = gray.astype(np.float32)\n    \n    # 1. Color statistics (per channel)\n    if patch.ndim == 3:\n        for channel in range(3):\n            ch = patch[:, :, channel].astype(np.float32)\n            features.extend([\n                np.mean(ch),\n                np.std(ch),\n                np.median(ch),\n                np.percentile(ch, 25),\n                np.percentile(ch, 75),\n                np.min(ch),\n                np.max(ch)\n            ])\n    \n    # 2. Grayscale statistics\n    features.extend([\n        np.mean(gray),\n        np.std(gray),\n        np.median(gray),\n        np.percentile(gray, 25),\n        np.percentile(gray, 75),\n        np.min(gray),\n        np.max(gray),\n        np.var(gray)\n    ])\n    \n    # 3. DCT features\n    dct_result = dct(dct(gray.T, norm='ortho').T, norm='ortho')\n    dct_feat = dct_result[:8, :8].flatten()\n    features.extend(dct_feat)\n    \n    # 4. Gradient features\n    grad_x = cv2.Sobel(gray, cv2.CV_32F, 1, 0, ksize=3)\n    grad_y = cv2.Sobel(gray, cv2.CV_32F, 0, 1, ksize=3)\n    grad_mag = np.sqrt(grad_x**2 + grad_y**2)\n    \n    features.extend([\n        np.mean(np.abs(grad_x)),\n        np.std(np.abs(grad_x)),\n        np.mean(np.abs(grad_y)),\n        np.std(np.abs(grad_y)),\n        np.mean(grad_mag),\n        np.std(grad_mag),\n        np.max(grad_mag)\n    ])\n    \n    # 5. Edge statistics\n    edges = cv2.Canny(gray.astype(np.uint8), 50, 150)\n    edge_density = np.sum(edges > 0) / edges.size\n    features.append(edge_density)\n    \n    # 6. Texture - Local Binary Pattern\n    lbp = local_binary_pattern(gray, 8, 1, method='uniform')\n    lbp_hist, _ = np.histogram(lbp.ravel(), bins=10, range=(0, 10), density=True)\n    features.extend(lbp_hist)\n    \n    # 7. Noise estimation\n    denoised = median_filter(gray, size=3)\n    noise = gray - denoised\n    features.extend([\n        np.std(noise),\n        np.mean(np.abs(noise)),\n        np.percentile(np.abs(noise), 95)\n    ])\n    \n    # 8. Frequency domain features\n    h, w = gray.shape\n    center_h, center_w = h // 2, w // 2\n    low_freq = np.sum(np.abs(dct_result[center_h-4:center_h+4, center_w-4:center_w+4]))\n    mid_freq = np.sum(np.abs(dct_result[center_h-8:center_h+8, center_w-8:center_w+8])) - low_freq\n    high_freq = np.sum(np.abs(dct_result)) - mid_freq - low_freq\n    total = low_freq + mid_freq + high_freq + 1e-10\n    features.extend([low_freq/total, mid_freq/total, high_freq/total])\n    \n    # 9. Texture variance in subregions\n    h_half, w_half = h // 2, w // 2\n    if h_half > 0 and w_half > 0:\n        subregions = [\n            gray[:h_half, :w_half],\n            gray[:h_half, w_half:],\n            gray[h_half:, :w_half],\n            gray[h_half:, w_half:]\n        ]\n        for region in subregions:\n            if region.size > 0:\n                features.append(np.var(region))\n    \n    # 10. HOG features (compact version)\n    if gray.shape[0] >= 16 and gray.shape[1] >= 16:\n        try:\n            hog_features = hog(gray, orientations=8, pixels_per_cell=(8, 8),\n                              cells_per_block=(1, 1), visualize=False, feature_vector=True)\n            # Take only first 16 HOG features for compactness\n            features.extend(hog_features[:16])\n        except:\n            # If HOG fails, add zeros\n            features.extend([0] * 16)\n    \n    return np.array(features)\n\n# ============================================================================\n# TRAINING DATA GENERATION\n# ============================================================================\n\ndef generate_training_data(authentic_images, forged_images, mask_files, config):\n    \"\"\"Generate training data from authentic and forged images\"\"\"\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"GENERATING TRAINING DATA\")\n    print(\"=\"*80)\n    \n    X_train = []\n    y_train = []\n    \n    # Sample authentic patches (label = 0)\n    print(\"\\n📊 Extracting authentic patches...\")\n    sample_size = min(len(authentic_images), config.MAX_TRAIN_IMAGES)\n    \n    for img_path in tqdm(authentic_images[:sample_size], desc=\"Authentic\"):\n        try:\n            img = cv2.imread(str(img_path))\n            if img is None:\n                continue\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            \n            h, w = img.shape[:2]\n            \n            # CRITICAL FIX: Check if image is large enough\n            if h < config.PATCH_SIZE or w < config.PATCH_SIZE:\n                continue\n            \n            # Random sampling of patches\n            patches_added = 0\n            max_attempts = config.PATCHES_PER_IMAGE * 3  # Allow more attempts\n            \n            for attempt in range(max_attempts):\n                if patches_added >= config.PATCHES_PER_IMAGE // 2:\n                    break\n                \n                y = np.random.randint(0, h - config.PATCH_SIZE + 1)\n                x = np.random.randint(0, w - config.PATCH_SIZE + 1)\n                \n                patch = img[y:y+config.PATCH_SIZE, x:x+config.PATCH_SIZE]\n                \n                # Validate patch dimensions\n                if patch.shape[0] != config.PATCH_SIZE or patch.shape[1] != config.PATCH_SIZE:\n                    continue\n                \n                features = extract_patch_features(patch)\n                \n                if features is not None:\n                    X_train.append(features)\n                    y_train.append(0)  # Authentic\n                    patches_added += 1\n        except Exception as e:\n            print(f\"Error processing {img_path.name}: {e}\")\n            continue\n    \n    # Sample forged patches (label = 1)\n    print(\"\\n📊 Extracting forged patches...\")\n    sample_size = min(len(forged_images), config.MAX_TRAIN_IMAGES)\n    \n    for img_path in tqdm(forged_images[:sample_size], desc=\"Forged\"):\n        try:\n            img_id = img_path.stem\n            \n            img = cv2.imread(str(img_path))\n            if img is None:\n                continue\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            \n            h, w = img.shape[:2]\n            \n            # CRITICAL FIX: Check if image is large enough\n            if h < config.PATCH_SIZE or w < config.PATCH_SIZE:\n                continue\n            \n            # Load mask if available\n            if img_id in mask_files:\n                mask = np.load(str(mask_files[img_id]))\n                \n                # Ensure mask is 2D\n                if mask.ndim > 2:\n                    mask = mask[:, :, 0] if mask.shape[2] == 1 else mask.max(axis=2)\n                \n                # Resize mask if needed\n                if mask.shape[:2] != (h, w):\n                    mask = cv2.resize(mask, (w, h), interpolation=cv2.INTER_NEAREST)\n                \n                mask = (mask > 0).astype(np.uint8)\n                \n                # Sample patches from forged regions\n                forged_coords = np.argwhere(mask > 0)\n                \n                if len(forged_coords) > 0:\n                    patches_added = 0\n                    max_attempts = config.PATCHES_PER_IMAGE * 3\n                    \n                    for attempt in range(max_attempts):\n                        if patches_added >= config.PATCHES_PER_IMAGE:\n                            break\n                        \n                        if len(forged_coords) == 0:\n                            break\n                        \n                        coord = forged_coords[np.random.randint(len(forged_coords))]\n                        y_center, x_center = coord\n                        \n                        y = max(0, min(y_center - config.PATCH_SIZE // 2, h - config.PATCH_SIZE))\n                        x = max(0, min(x_center - config.PATCH_SIZE // 2, w - config.PATCH_SIZE))\n                        \n                        # Validate bounds\n                        if y < 0 or x < 0 or y + config.PATCH_SIZE > h or x + config.PATCH_SIZE > w:\n                            continue\n                        \n                        patch = img[y:y+config.PATCH_SIZE, x:x+config.PATCH_SIZE]\n                        \n                        # Validate patch dimensions\n                        if patch.shape[0] != config.PATCH_SIZE or patch.shape[1] != config.PATCH_SIZE:\n                            continue\n                        \n                        features = extract_patch_features(patch)\n                        \n                        if features is not None:\n                            X_train.append(features)\n                            y_train.append(1)  # Forged\n                            patches_added += 1\n            \n            # Also sample some non-forged patches from forged images\n            patches_added = 0\n            max_attempts = config.PATCHES_PER_IMAGE\n            \n            for attempt in range(max_attempts):\n                if patches_added >= config.PATCHES_PER_IMAGE // 4:\n                    break\n                \n                y = np.random.randint(0, h - config.PATCH_SIZE + 1)\n                x = np.random.randint(0, w - config.PATCH_SIZE + 1)\n                \n                patch = img[y:y+config.PATCH_SIZE, x:x+config.PATCH_SIZE]\n                \n                # Validate patch dimensions\n                if patch.shape[0] != config.PATCH_SIZE or patch.shape[1] != config.PATCH_SIZE:\n                    continue\n                \n                # Check if patch overlaps with forgery\n                if img_id in mask_files:\n                    patch_mask = mask[y:y+config.PATCH_SIZE, x:x+config.PATCH_SIZE]\n                    if patch_mask.sum() / patch_mask.size < 0.1:  # Less than 10% forged\n                        features = extract_patch_features(patch)\n                        if features is not None:\n                            X_train.append(features)\n                            y_train.append(0)  # Authentic\n                            patches_added += 1\n                else:\n                    features = extract_patch_features(patch)\n                    if features is not None:\n                        X_train.append(features)\n                        y_train.append(0)  # Authentic\n                        patches_added += 1\n        \n        except Exception as e:\n            print(f\"Error processing {img_path.name}: {e}\")\n            continue\n    \n    X_train = np.array(X_train)\n    y_train = np.array(y_train)\n    \n    # Limit dataset size\n    if len(X_train) > config.MAX_TRAIN_SAMPLES:\n        indices = np.random.choice(len(X_train), config.MAX_TRAIN_SAMPLES, replace=False)\n        X_train = X_train[indices]\n        y_train = y_train[indices]\n    \n    print(f\"\\n✓ Training samples: {len(X_train)}\")\n    print(f\"  - Authentic: {np.sum(y_train == 0)}\")\n    print(f\"  - Forged: {np.sum(y_train == 1)}\")\n    print(f\"  - Feature dimensions: {X_train.shape[1]}\")\n    \n    return X_train, y_train\n\n# ============================================================================\n# MODEL TRAINING\n# ============================================================================\n\ndef train_model(X_train, y_train, config):\n    \"\"\"Train GPU-accelerated XGBoost model\"\"\"\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"TRAINING MODEL\")\n    print(\"=\"*80)\n    \n    # Split data\n    X_tr, X_val, y_tr, y_val = train_test_split(X_train, y_train, test_size=0.15, random_state=42)\n    \n    # Standardize features\n    scaler = StandardScaler()\n    X_tr = scaler.fit_transform(X_tr)\n    X_val = scaler.transform(X_val)\n    \n    print(f\"\\n📊 Training set: {len(X_tr)}, Validation set: {len(X_val)}\")\n    \n    # Train XGBoost with GPU\n    print(\"\\n🚀 Training XGBoost with GPU acceleration...\")\n    \n    params = {\n        'tree_method': 'hist',\n        'device': 'cuda' if config.USE_GPU else 'cpu',\n        'max_depth': config.MAX_DEPTH,\n        'learning_rate': config.LEARNING_RATE,\n        'n_estimators': config.N_ESTIMATORS,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'objective': 'binary:logistic',\n        'eval_metric': 'logloss',\n        'random_state': 42\n    }\n    \n    model = xgb.XGBClassifier(**params)\n    \n    model.fit(\n        X_tr, y_tr,\n        eval_set=[(X_val, y_val)],\n        verbose=50\n    )\n    \n    # Evaluate\n    train_acc = model.score(X_tr, y_tr)\n    val_acc = model.score(X_val, y_val)\n    \n    print(f\"\\n✓ Training accuracy: {train_acc:.4f}\")\n    print(f\"✓ Validation accuracy: {val_acc:.4f}\")\n    \n    return model, scaler\n\n# ============================================================================\n# PREDICTION AND SEGMENTATION\n# ============================================================================\n\ndef predict_image(image, model, scaler, config):\n    \"\"\"Predict forgery mask for an image using sliding window\"\"\"\n    \n    h, w = image.shape[:2]\n    \n    # Create prediction map\n    prediction_map = np.zeros((h, w), dtype=np.float32)\n    count_map = np.zeros((h, w), dtype=np.int32)\n    \n    patch_size = config.PATCH_SIZE\n    stride = config.STRIDE\n    \n    # Check if image is large enough\n    if h < patch_size or w < patch_size:\n        return np.zeros((h, w), dtype=np.uint8), 0.0\n    \n    # Sliding window\n    patches = []\n    positions = []\n    \n    for y in range(0, h - patch_size + 1, stride):\n        for x in range(0, w - patch_size + 1, stride):\n            patch = image[y:y+patch_size, x:x+patch_size]\n            \n            # Validate patch\n            if patch.shape[0] != patch_size or patch.shape[1] != patch_size:\n                continue\n            \n            features = extract_patch_features(patch)\n            \n            if features is not None:\n                patches.append(features)\n                positions.append((y, x))\n    \n    if len(patches) == 0:\n        return np.zeros((h, w), dtype=np.uint8), 0.0\n    \n    # Batch prediction\n    X = np.array(patches)\n    X = scaler.transform(X)\n    \n    predictions = model.predict_proba(X)[:, 1]  # Probability of forgery\n    \n    # Aggregate predictions\n    for (y, x), pred in zip(positions, predictions):\n        prediction_map[y:y+patch_size, x:x+patch_size] += pred\n        count_map[y:y+patch_size, x:x+patch_size] += 1\n    \n    # Average predictions\n    mask = np.divide(prediction_map, count_map, where=count_map > 0)\n    \n    # Calculate confidence\n    confidence = np.mean(predictions)\n    \n    # Threshold\n    binary_mask = (mask > config.FORGERY_THRESHOLD).astype(np.uint8)\n    \n    # Morphological operations\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))\n    binary_mask = cv2.morphologyEx(binary_mask, cv2.MORPH_CLOSE, kernel, iterations=2)\n    binary_mask = cv2.morphologyEx(binary_mask, cv2.MORPH_OPEN, kernel)\n    \n    # Remove small regions\n    labeled = label(binary_mask)\n    regions = regionprops(labeled)\n    \n    refined_mask = np.zeros_like(binary_mask)\n    for region in regions:\n        if region.area >= config.MIN_REGION_AREA:\n            coords = region.coords\n            refined_mask[coords[:, 0], coords[:, 1]] = 1\n    \n    return refined_mask, confidence\n\n# ============================================================================\n# VISUALIZATION\n# ============================================================================\n\ndef visualize_detection(image, mask, case_id, confidence, save_path='detection_viz.png'):\n    \"\"\"Visualize detection results\"\"\"\n    fig, axes = plt.subplots(1, 3, figsize=(15, 5))\n    \n    axes[0].imshow(image)\n    axes[0].set_title(f'Original (ID: {case_id})')\n    axes[0].axis('off')\n    \n    axes[1].imshow(mask, cmap='hot')\n    axes[1].set_title(f'Prediction')\n    axes[1].axis('off')\n    \n    overlay = image.copy()\n    if mask.max() > 0:\n        mask_colored = np.zeros_like(image)\n        mask_colored[:, :, 0] = mask * 255\n        overlay = cv2.addWeighted(overlay, 0.7, mask_colored, 0.3, 0)\n    \n    axes[2].imshow(overlay)\n    axes[2].set_title(f'Overlay (Conf: {confidence:.3f})')\n    axes[2].axis('off')\n    \n    plt.tight_layout()\n    plt.savefig(save_path, dpi=100, bbox_inches='tight')\n    plt.close()\n    \n    print(f\"   💾 Saved: {save_path}\")\n\n# ============================================================================\n# RLE ENCODING\n# ============================================================================\n\ndef rle_encode(mask):\n    \"\"\"Run-length encode binary mask\"\"\"\n    dots = np.where(mask.T.flatten() == 1)[0]\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\n# ============================================================================\n# MAIN\n# ============================================================================\n\ndef main():\n    config = Config()\n    \n    # Discover data\n    authentic_images, forged_images, mask_files, test_images = discover_data(config)\n    \n    # Generate training data\n    X_train, y_train = generate_training_data(authentic_images, forged_images, mask_files, config)\n    \n    # Ensure we have training data\n    if len(X_train) == 0:\n        print(\"\\n❌ ERROR: No training data generated!\")\n        return None\n    \n    # Train model\n    model, scaler = train_model(X_train, y_train, config)\n    \n    # Save model\n    joblib.dump(model, 'forgery_detector.pkl')\n    joblib.dump(scaler, 'feature_scaler.pkl')\n    print(\"\\n✓ Model saved\")\n    \n    # Predict on test images\n    print(\"\\n\" + \"=\"*80)\n    print(\"PREDICTING ON TEST IMAGES\")\n    print(\"=\"*80)\n    \n    sample_sub = pd.read_csv(config.SAMPLE_SUB_PATH)\n    results = []\n    \n    viz_count = 0\n    detection_summary = {'authentic': 0, 'forgery': 0, 'confidences': []}\n    \n    for idx, row in tqdm(sample_sub.iterrows(), total=len(sample_sub), desc=\"🔍\"):\n        case_id = str(row['case_id'])\n        \n        test_img_path = None\n        for img_path in test_images:\n            if img_path.stem == case_id:\n                test_img_path = img_path\n                break\n        \n        if test_img_path and test_img_path.exists():\n            try:\n                img = cv2.imread(str(test_img_path))\n                img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n                \n                print(f\"\\n📷 {case_id}: {img.shape[1]}x{img.shape[0]}px\")\n                \n                mask, confidence = predict_image(img, model, scaler, config)\n                \n                detection_summary['confidences'].append(confidence)\n                \n                is_forgery = confidence >= 0.5 and mask.sum() >= config.MIN_REGION_AREA\n                \n                if not is_forgery:\n                    results.append({'case_id': int(case_id), 'annotation': 'authentic'})\n                    detection_summary['authentic'] += 1\n                    print(f\"   ✓ AUTHENTIC ({confidence:.3f})\")\n                else:\n                    run_lengths = rle_encode(mask)\n                    \n                    if len(run_lengths) > 0:\n                        results.append({'case_id': int(case_id), 'annotation': json.dumps([int(x) for x in run_lengths])})\n                        detection_summary['forgery'] += 1\n                        print(f\"   ⚠️  FORGERY ({confidence:.3f})\")\n                    else:\n                        results.append({'case_id': int(case_id), 'annotation': 'authentic'})\n                        detection_summary['authentic'] += 1\n                \n                if config.VISUALIZE_SAMPLES and viz_count < config.MAX_VIZ_SAMPLES:\n                    visualize_detection(img, mask, case_id, confidence, f'detection_{case_id}.png')\n                    viz_count += 1\n            \n            except Exception as e:\n                print(f\"Error processing test image {case_id}: {e}\")\n                results.append({'case_id': int(case_id), 'annotation': 'authentic'})\n                detection_summary['authentic'] += 1\n        else:\n            results.append({'case_id': int(case_id), 'annotation': 'authentic'})\n            detection_summary['authentic'] += 1\n    \n    submission_df = pd.DataFrame(results)\n    submission_df.to_csv('submission.csv', index=False)\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"COMPLETE\")\n    print(\"=\"*80)\n    print(f\"✓ Total: {len(submission_df)}\")\n    print(f\"✓ Authentic: {detection_summary['authentic']}\")\n    print(f\"✓ Forgeries: {detection_summary['forgery']}\")\n    print(f\"💾 Saved: submission.csv\")\n    print(\"=\"*80)\n    \n    return submission_df\n\nif __name__ == \"__main__\":\n    submission = main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}