{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":113558,"databundleVersionId":14456136,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":676295,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":512741,"modelId":527383}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🔬 Scientific Image Forgery Detection - Inference","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom glob import glob\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.models as models\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nprint(f\"PyTorch: {torch.__version__}\")\nprint(f\"CUDA: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:01:57.457230Z","iopub.execute_input":"2025-12-09T06:01:57.457907Z","iopub.status.idle":"2025-12-09T06:02:37.746396Z","shell.execute_reply.started":"2025-12-09T06:01:57.457880Z","shell.execute_reply":"2025-12-09T06:02:37.745650Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ⚙️ Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    # Data paths\n    BASE_PATH = '/kaggle/input/recodai-luc-scientific-image-forgery-detection'\n    TEST_IMAGES = f'{BASE_PATH}/test_images'\n    \n    # Model path - UPDATE THIS to your uploaded model\n    MODEL_PATH = '/kaggle/input/resnet34-recod-ai/pytorch/default/1/best_model.pth'\n    \n    # Model config (MUST match training)\n    ENCODER = 'resnet34'\n    IMG_SIZE = 512\n    \n    # Inference settings\n    THRESHOLD = 0.09  # Tuned threshold\n    MIN_COMPONENT_SIZE = 30\n    MAX_COMPONENT_RATIO = 0.3\n    \n    USE_TTA = True\n    TTA_AUGMENTS = 4\n    \n    DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    MIXED_PRECISION = True\n\nprint(f\"Device: {CFG.DEVICE}\")\nprint(f\"Model path: {CFG.MODEL_PATH}\")\nprint(f\"Model exists: {os.path.exists(CFG.MODEL_PATH)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:37.747760Z","iopub.execute_input":"2025-12-09T06:02:37.748122Z","iopub.status.idle":"2025-12-09T06:02:37.754190Z","shell.execute_reply.started":"2025-12-09T06:02:37.748102Z","shell.execute_reply":"2025-12-09T06:02:37.753582Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## RLE Encoding","metadata":{}},{"cell_type":"code","source":"def rle_encode(mask):\n    \"\"\"Run-length encode a binary mask.\"\"\"\n    pixels = mask.flatten(order='F')\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0]\n    \n    if len(runs) == 0:\n        return \"authentic\"\n    \n    runs[1::2] -= runs[::2]\n    runs[::2] += 1\n    \n    return \"[\" + \" \".join(str(x) for x in runs) + \"]\"\n\n# Test\ntest_mask = np.zeros((100, 100), dtype=np.uint8)\ntest_mask[20:40, 30:60] = 1\nprint(f\"✓ RLE test: {rle_encode(test_mask)[:30]}...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:37.754978Z","iopub.execute_input":"2025-12-09T06:02:37.755552Z","iopub.status.idle":"2025-12-09T06:02:37.772038Z","shell.execute_reply.started":"2025-12-09T06:02:37.755533Z","shell.execute_reply":"2025-12-09T06:02:37.771446Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🏗️ Model Architecture","metadata":{}},{"cell_type":"code","source":"class ConvBlock(nn.Module):\n    def __init__(self, in_ch, out_ch):\n        super().__init__()\n        self.conv = nn.Sequential(\n            nn.Conv2d(in_ch, out_ch, 3, padding=1, bias=False),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_ch, out_ch, 3, padding=1, bias=False),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True),\n        )\n    \n    def forward(self, x):\n        return self.conv(x)\n\n\nclass UNetDecoder(nn.Module):\n    def __init__(self, encoder_channels, decoder_channels):\n        super().__init__()\n        \n        self.up_convs = nn.ModuleList()\n        self.dec_convs = nn.ModuleList()\n        \n        in_ch = encoder_channels[-1]\n        \n        for i, out_ch in enumerate(decoder_channels):\n            skip_ch = encoder_channels[-(i+2)] if i < len(encoder_channels) - 1 else 0\n            \n            self.up_convs.append(\n                nn.ConvTranspose2d(in_ch, out_ch, kernel_size=2, stride=2)\n            )\n            self.dec_convs.append(\n                ConvBlock(out_ch + skip_ch, out_ch)\n            )\n            in_ch = out_ch\n    \n    def forward(self, features):\n        x = features[-1]\n        \n        for i, (up, dec) in enumerate(zip(self.up_convs, self.dec_convs)):\n            x = up(x)\n            \n            if i < len(features) - 1:\n                skip = features[-(i+2)]\n                if x.shape[2:] != skip.shape[2:]:\n                    x = F.interpolate(x, size=skip.shape[2:], mode='bilinear', align_corners=False)\n                x = torch.cat([x, skip], dim=1)\n            \n            x = dec(x)\n        \n        return x\n\n\n\n\nclass ResNetEncoder(nn.Module):\n    def __init__(self, name='resnet34'):\n        super().__init__()\n        \n        if name == 'resnet34':\n            resnet = models.resnet34(weights=None)\n            self.channels = [64, 64, 128, 256, 512]\n        elif name == 'resnet18':\n            resnet = models.resnet18(weights=None)\n            self.channels = [64, 64, 128, 256, 512]\n        elif name == 'resnet50':\n            resnet = models.resnet50(weights=None)\n            self.channels = [64, 256, 512, 1024, 2048]\n        else:\n            raise ValueError(f\"Unknown encoder: {name}\")\n        \n        self.stage0 = nn.Sequential(resnet.conv1, resnet.bn1, resnet.relu)\n        self.pool = resnet.maxpool\n        self.stage1 = resnet.layer1\n        self.stage2 = resnet.layer2\n        self.stage3 = resnet.layer3\n        self.stage4 = resnet.layer4\n    \n    def forward(self, x):\n        features = []\n        \n        x = self.stage0(x)\n        features.append(x)\n        \n        x = self.pool(x)\n        x = self.stage1(x)\n        features.append(x)\n        \n        x = self.stage2(x)\n        features.append(x)\n        \n        x = self.stage3(x)\n        features.append(x)\n        \n        x = self.stage4(x)\n        features.append(x)\n        \n        return features\n\n\n\nclass UNet(nn.Module):\n    def __init__(self, encoder_name='resnet34', num_classes=1):\n        super().__init__()\n        \n        self.encoder = ResNetEncoder(encoder_name)\n        \n        enc_channels = self.encoder.channels\n        dec_channels = [256, 128, 64, 32]\n        \n        self.decoder = UNetDecoder(enc_channels, dec_channels)\n        \n        self.final_up = nn.ConvTranspose2d(32, 32, kernel_size=2, stride=2)\n        self.final_conv = nn.Conv2d(32, num_classes, kernel_size=1)\n    \n    def forward(self, x):\n        features = self.encoder(x)\n        x = self.decoder(features)\n        x = self.final_up(x)\n        x = self.final_conv(x)\n        return x\n\nprint(\"✓ Model architecture defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:37.772770Z","iopub.execute_input":"2025-12-09T06:02:37.773018Z","iopub.status.idle":"2025-12-09T06:02:37.789210Z","shell.execute_reply.started":"2025-12-09T06:02:37.772994Z","shell.execute_reply":"2025-12-09T06:02:37.788644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📦 Load Trained Model","metadata":{}},{"cell_type":"code","source":"def load_model():\n    model = UNet(encoder_name=CFG.ENCODER)\n    \n    state_dict = torch.load(CFG.MODEL_PATH, map_location=CFG.DEVICE, weights_only=False)\n    model.load_state_dict(state_dict)\n    \n    model = model.to(CFG.DEVICE)\n    model.eval()\n    \n    print(f\"✓ Model loaded from {CFG.MODEL_PATH}\")\n    print(f\"✓ Parameters: {sum(p.numel() for p in model.parameters()):,}\")\n    \n    return model\n\n\n\n\n\nmodel = load_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:37.791040Z","iopub.execute_input":"2025-12-09T06:02:37.791236Z","iopub.status.idle":"2025-12-09T06:02:39.801674Z","shell.execute_reply.started":"2025-12-09T06:02:37.791220Z","shell.execute_reply":"2025-12-09T06:02:39.800818Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🔧 Transforms & Post-processing","metadata":{}},{"cell_type":"code","source":"def get_transforms():\n    return A.Compose([\n        A.Resize(CFG.IMG_SIZE, CFG.IMG_SIZE),\n        A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n        ToTensorV2(),\n    ])\ndef postprocess_mask(mask, threshold, min_size, max_ratio):\n    \"\"\"Remove small noise and oversized regions.\"\"\"\n    binary = (mask > threshold).astype(np.uint8)\n    \n    if binary.sum() == 0:\n        return binary\n    \n    num_labels, labels, stats, _ = cv2.connectedComponentsWithStats(binary, connectivity=8)\n    \n    total_pixels = mask.shape[0] * mask.shape[1]\n    output = np.zeros_like(binary)\n    \n    for i in range(1, num_labels):\n        area = stats[i, cv2.CC_STAT_AREA]\n        if area >= min_size and area < total_pixels * max_ratio:\n            output[labels == i] = 1\n    \n    return output\n\n\n\nprint(\"✓ Transforms defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:39.802555Z","iopub.execute_input":"2025-12-09T06:02:39.802823Z","iopub.status.idle":"2025-12-09T06:02:39.809240Z","shell.execute_reply.started":"2025-12-09T06:02:39.802803Z","shell.execute_reply":"2025-12-09T06:02:39.808518Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🔮 Inference Functions","metadata":{}},{"cell_type":"code","source":"def predict_single(model, image, transform):\n    \"\"\"Predict without TTA.\"\"\"\n    augmented = transform(image=image)\n    img_tensor = augmented['image'].unsqueeze(0).to(CFG.DEVICE)\n    \n    with torch.no_grad():\n        with torch.cuda.amp.autocast(enabled=CFG.MIXED_PRECISION):\n            output = model(img_tensor)\n    \n    pred =   torch.sigmoid(output).cpu().numpy()[0, 0]\n    return pred.astype(np.float32)  \n\n\n\ndef predict_with_tta(model, image, transform, n_tta=4):\n    \"\"\"Predict with test-time augmentation.\"\"\"\n    augments = [\n        lambda x: x,\n        lambda x: np.fliplr(x).copy(),\n        lambda x: np.flipud(x).copy(),\n        lambda x: np.rot90(x, 1).copy(),\n    ]\n    \n    predictions = []\n    \n    for i, aug_fn in enumerate(augments[:n_tta]):\n        aug_img = aug_fn(image)\n        pred = predict_single(model, aug_img, transform)\n        \n        # Reverse augmentation\n        if i == 1:\n            pred = np.fliplr(pred).copy()\n        elif i == 2:\n            pred = np.flipud(pred).copy()\n        elif i == 3:\n            pred = np.rot90(pred, -1).copy()\n        \n        predictions.append(pred)\n    \n    avg_pred = np.mean(predictions, axis=0)\n    return avg_pred.astype(np.float32)  # Ensure float32\n\n\ndef predict_image(model, image_path, transform):\n    \"\"\"Full prediction pipeline for one image.\"\"\"\n    # Load image\n    image = cv2.imread(image_path)\n    original_shape = image.shape[:2]  # (H, W)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Predict\n    if CFG.USE_TTA:\n        pred = predict_with_tta(model, image, transform, CFG.TTA_AUGMENTS)\n    else:\n        pred = predict_single(model, image, transform)\n    \n    # Ensure pred is numpy array with correct dtype\n    if isinstance(pred, torch.Tensor):\n        pred = pred.cpu().numpy()\n    pred = pred.astype(np.float32)\n    \n    # Resize to original size\n    pred = cv2.resize(pred, (original_shape[1], original_shape[0]))\n    \n    # Post-process\n    mask = postprocess_mask(pred, CFG.THRESHOLD, CFG.MIN_COMPONENT_SIZE, CFG.MAX_COMPONENT_RATIO)\n    \n    return mask\nprint(\"✓ Inference functions defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:39.809903Z","iopub.execute_input":"2025-12-09T06:02:39.810191Z","iopub.status.idle":"2025-12-09T06:02:39.825027Z","shell.execute_reply.started":"2025-12-09T06:02:39.810173Z","shell.execute_reply":"2025-12-09T06:02:39.824322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📊 Explore Test Data","metadata":{}},{"cell_type":"code","source":"# Check test data structure\nprint(\"=== Test Data Structure ===\")\nprint(f\"Test path: {CFG.TEST_IMAGES}\")\nprint(f\"Exists: {os.path.exists(CFG.TEST_IMAGES)}\")\n\nif os.path.exists(CFG.TEST_IMAGES):\n    contents = os.listdir(CFG.TEST_IMAGES)\n    print(f\"Contents ({len(contents)} items): {contents[:10]}\")\n    \n    # Count files\n    png_files = glob(f\"{CFG.TEST_IMAGES}/*.png\")\n    jpg_files = glob(f\"{CFG.TEST_IMAGES}/*.jpg\")\n    print(f\"\\nPNG files: {len(png_files)}\")\n    print(f\"JPG files: {len(jpg_files)}\")\n    \n    # Check for subfolders\n    for item in contents[:5]:\n        item_path = os.path.join(CFG.TEST_IMAGES, item)\n        if os.path.isdir(item_path):\n            print(f\"\\nSubfolder '{item}': {len(os.listdir(item_path))} files\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:02:39.825841Z","iopub.execute_input":"2025-12-09T06:02:39.826089Z","iopub.status.idle":"2025-12-09T06:02:39.848690Z","shell.execute_reply.started":"2025-12-09T06:02:39.826066Z","shell.execute_reply":"2025-12-09T06:02:39.848092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📤 Create Submission","metadata":{}},{"cell_type":"code","source":"def get_test_images():\n    \"\"\"Get all test image paths.\"\"\"\n    test_images = []\n    \n    # Direct files in test_images/\n    test_images.extend(glob(f\"{CFG.TEST_IMAGES}/*.png\"))\n    test_images.extend(glob(f\"{CFG.TEST_IMAGES}/*.jpg\"))\n    \n    # Check for subfolders (in case structure is like train)\n    for subfolder in ['authentic', 'forged', '']:\n        subfolder_path = os.path.join(CFG.TEST_IMAGES, subfolder)\n        if os.path.exists(subfolder_path) and os.path.isdir(subfolder_path):\n            test_images.extend(glob(f\"{subfolder_path}/*.png\"))\n            test_images.extend(glob(f\"{subfolder_path}/*.jpg\"))\n    \n    # Remove duplicates\n    test_images = list(set(test_images))\n    \n    return sorted(test_images)\n\nimport json\n\ndef rle_encode(mask):\n    \"\"\"RLE encode - returns JSON array string.\"\"\"\n    pixels = mask.T.flatten()\n    dots = np.where(pixels == 1)[0]\n    \n    if len(dots) == 0:\n        return \"authentic\"\n    \n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    \n    return json.dumps([int(x) for x in run_lengths])\n\n\ndef postprocess(preds, original_size, alpha_grad=0.35):\n    \"\"\"Enhanced postprocessing with edge detection.\"\"\"\n    gx = cv2.Sobel(preds, cv2.CV_32F, 1, 0, ksize=3)\n    gy = cv2.Sobel(preds, cv2.CV_32F, 0, 1, ksize=3)\n    grad_mag = np.sqrt(gx**2 + gy**2)\n    grad_norm = grad_mag / (grad_mag.max() + 1e-6)\n    enhanced = (1 - alpha_grad) * preds + alpha_grad * grad_norm\n    enhanced = cv2.GaussianBlur(enhanced, (3, 3), 0)\n    thr = np.mean(enhanced) + 0.3 * np.std(enhanced)\n    mask = (enhanced > thr).astype(np.uint8)\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5, 5), np.uint8))\n    mask = cv2.morphologyEx(mask, cv2.MORPH_OPEN, np.ones((3, 3), np.uint8))\n    mask = cv2.resize(mask, original_size, interpolation=cv2.INTER_NEAREST)\n    return mask\n\n\ndef infer_image(model, image_path, transform):\n    \"\"\"Infer single image with filtering.\"\"\"\n    # Load image\n    image = Image.open(image_path).convert(\"RGB\")\n    original_size = image.size  # (W, H)\n    \n    # Preprocess\n    image_resized = image.resize((CFG.IMG_SIZE, CFG.IMG_SIZE))\n    image_array = np.array(image_resized, np.float32) / 255\n    image_tensor = torch.from_numpy(image_array).permute(2, 0, 1)[None].to(CFG.DEVICE)\n    \n    # Predict\n    with torch.no_grad():\n        output = model(image_tensor)\n        preds = torch.sigmoid(output)[0, 0].cpu().numpy()\n    \n    # Postprocess\n    mask = postprocess(preds, original_size)\n    \n    # Calculate area and mean confidence\n    area = int(mask.sum())\n    if area > 0:\n        mask_resized = cv2.resize(mask, (CFG.IMG_SIZE, CFG.IMG_SIZE), interpolation=cv2.INTER_NEAREST)\n        mean_inside = float(preds[mask_resized == 1].mean())\n    else:\n        mean_inside = 0.0\n    \n    # Filter: if too small or low confidence, mark as authentic\n    if area < 400 or mean_inside < 0.3:\n        return \"authentic\", None\n    \n    return \"forged\", mask\n\n\ndef create_submission(model):\n    \"\"\"Create submission file.\"\"\"\n    from PIL import Image\n    \n    test_images = sorted(glob(f\"{CFG.TEST_IMAGES}/*.png\") + glob(f\"{CFG.TEST_IMAGES}/*.jpg\"))\n    print(f\"Test images: {len(test_images)}\")\n    \n    predictions = []\n    \n    for image_path in tqdm(test_images, desc=\"Running Inference\"):\n        case_id = Path(image_path).stem\n        \n        label, mask = infer_image(model, image_path, None)\n        \n        if label == \"authentic\":\n            annotation = \"authentic\"\n        else:\n            annotation = rle_encode((mask > 0).astype(np.uint8))\n        \n        predictions.append({\n            \"case_id\": case_id,\n            \"annotation\": annotation\n        })\n    \n    # Create dataframe\n    predictions_df = pd.DataFrame(predictions)\n    predictions_df[\"case_id\"] = predictions_df[\"case_id\"].astype(str)\n    \n    # Merge with sample submission\n    sample_sub = pd.read_csv(f\"{CFG.BASE_PATH}/sample_submission.csv\")\n    sample_sub[\"case_id\"] = sample_sub[\"case_id\"].astype(str)\n    \n    submission = sample_sub[[\"case_id\"]].merge(predictions_df, on=\"case_id\", how=\"left\")\n    submission[\"annotation\"] = submission[\"annotation\"].fillna(\"authentic\")\n    \n    # Save\n    submission.to_csv(\"submission.csv\", index=False)\n    \n    print(f\"\\n✓ Submission saved!\")\n    print(submission)\n    \n    return submission\n\n\n# Run\nfrom PIL import Image\nsubmission = create_submission(model)\n\n# Run it\nsubmission = create_submission(model)\nprint(\"\\nSubmission preview:\")\nprint(submission)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:11:23.098980Z","iopub.execute_input":"2025-12-09T06:11:23.099633Z","iopub.status.idle":"2025-12-09T06:11:23.372444Z","shell.execute_reply.started":"2025-12-09T06:11:23.099583Z","shell.execute_reply":"2025-12-09T06:11:23.371868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 👀 Visualize Predictions","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef visualize_predictions(model, n=6):\n    \n    test_images = get_test_images()[:n]\n    transform = get_transforms()\n    \n    fig, axes = plt.subplots(n, 3, figsize=(15, 5*n))\n    \n    for i, image_path in enumerate(test_images):\n        try:\n            # Load image\n            image = cv2.imread(image_path)\n            if image is None:\n                print(f\"Failed to load: {image_path}\")\n                continue\n            image_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            \n            # Predict\n            pred_mask = predict_image(model, image_path, transform)\n            \n            # Plot original\n            axes[i, 0].imshow(image_rgb)\n            axes[i, 0].set_title(f'{Path(image_path).name}')\n            axes[i, 0].axis('off')\n            \n            # Plot mask\n            axes[i, 1].imshow(pred_mask, cmap='hot')\n            axes[i, 1].set_title(f'Predicted ({pred_mask.sum()} pixels)')\n            axes[i, 1].axis('off')\n            \n            # Plot overlay\n            overlay = image_rgb.copy()\n            overlay[pred_mask > 0] = [255, 0, 0]\n            axes[i, 2].imshow(overlay)\n            axes[i, 2].set_title('Overlay')\n            axes[i, 2].axis('off')\n            \n        except Exception as e:\n            print(f\"Error with {image_path}: {e}\")\n    \n    plt.tight_layout()\n    plt.show()\n\nvisualize_predictions(model, n=6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:11:28.273836Z","iopub.execute_input":"2025-12-09T06:11:28.274626Z","iopub.status.idle":"2025-12-09T06:11:31.049359Z","shell.execute_reply.started":"2025-12-09T06:11:28.274581Z","shell.execute_reply":"2025-12-09T06:11:31.048650Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🔍 Check Submission","metadata":{}},{"cell_type":"code","source":"# Final check\nprint(\"=== Submission Summary ===\")\nprint(f\"Shape: {submission.shape}\")\nprint(f\"\\nFirst 10 rows:\")\nprint(submission.head(10))\nprint(f\"\\nLast 10 rows:\")\nprint(submission.tail(10))\nprint(f\"\\nAnnotation distribution:\")\nprint(f\"  Authentic: {(submission['annotation'] == 'authentic').sum()}\")\nprint(f\"  Forged: {(submission['annotation'] != 'authentic').sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-09T06:11:35.078187Z","iopub.execute_input":"2025-12-09T06:11:35.078699Z","iopub.status.idle":"2025-12-09T06:11:35.085891Z","shell.execute_reply.started":"2025-12-09T06:11:35.078675Z","shell.execute_reply":"2025-12-09T06:11:35.084963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}