{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":16880,"databundleVersionId":858837},{"sourceType":"datasetVersion","sourceId":15437194,"datasetId":9875486,"databundleVersionId":16356837}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"053983f8-22c3-42a3-8e9a-834f6b29dfa2","cell_type":"markdown","source":"# DeepFake Detection — Final Year Project\n\n### Solution Architecture\n| Component | Details |\n|---|---|\n| **Face Detector** | MTCNN (facenet-pytorch) |\n| **Backbone** | EfficientNet-B7 Noisy Student |\n| **Input Resolution** | 380 × 380 |\n| **Frames per Video** | 32 |\n| **Face Margin** | 30% around detected face |\n| **Aggregation** | Confident Strategy |\n| **Augmentations** | Albumentations (heavy) |\n| **Ensemble** | 5 models × different seeds |","metadata":{}},{"id":"aeaf32d8-ac97-440d-9b8f-7200fdb2cdfc","cell_type":"markdown","source":"## 1. Install Dependencies","metadata":{}},{"id":"6d14e3c7-ef4d-4df7-bc90-87d8a4e5fe20","cell_type":"code","source":"%%capture\n# Install compatible versions ONLY\n!pip install torch==1.10.2 torchvision==0.11.3 torchaudio==0.10.2\n\n!pip install timm==0.4.5\n!pip install albumentations==0.5.2\n!pip install opencv-python-headless==4.5.5.64\n!pip install facenet-pytorch==2.5.2\n!pip install scikit-learn pandas tqdm matplotlib","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:13:07.405616Z","iopub.execute_input":"2026-03-31T14:13:07.405874Z","iopub.status.idle":"2026-03-31T14:15:18.552254Z","shell.execute_reply.started":"2026-03-31T14:13:07.405835Z","shell.execute_reply":"2026-03-31T14:15:18.551369Z"}},"outputs":[],"execution_count":null},{"id":"3cce6b33-4557-422e-ab38-4e2f1483cbe7","cell_type":"markdown","source":"## 2. Imports","metadata":{}},{"id":"5f9d92c9-b502-4e9c-a535-f3ebe0304505","cell_type":"code","source":"import os\nimport cv2\nimport json\nimport random\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\n\n# ❌ NO AMP (torch 1.3 / 1.10 safe)\n\nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom facenet_pytorch import MTCNN\nfrom pathlib import Path\nfrom tqdm import tqdm   # safer than tqdm.notebook\nfrom sklearn.metrics import log_loss, roc_auc_score, roc_curve\nfrom sklearn.model_selection import GroupShuffleSplit\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    \n    if torch.cuda.is_available():\n        torch.cuda.manual_seed_all(seed)\n\nSEED = 42\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nseed_everything(SEED)\n\nprint(f\"Device : {DEVICE}\")\nprint(f\"Torch  : {torch.__version__}\")\n\nif torch.cuda.is_available():\n    print(f\"GPU    : {torch.cuda.get_device_name(0)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:18.554618Z","iopub.execute_input":"2026-03-31T14:15:18.555100Z","iopub.status.idle":"2026-03-31T14:15:21.583762Z","shell.execute_reply.started":"2026-03-31T14:15:18.554900Z","shell.execute_reply":"2026-03-31T14:15:21.582927Z"}},"outputs":[],"execution_count":null},{"id":"e4aef7f2-a4fc-4b61-aa64-a3f9d567b7a6","cell_type":"markdown","source":"## 3. Configuration","metadata":{}},{"id":"f235fb50-a344-4552-a868-a174e5009c82","cell_type":"code","source":"class CFG:\n    DATA_ROOT    = Path('/kaggle/input/deepfake-detection-challenge')\n    TRAIN_DIR    = DATA_ROOT / 'train_sample_videos'\n    TEST_DIR     = DATA_ROOT / 'test_videos'\n    CROPS_DIR    = Path('/kaggle/working/crops')\n    WEIGHTS_DIR  = Path('/kaggle/working/weights')\n\n    ENCODER      = 'tf_efficientnet_b7_ns'\n    INPUT_SIZE   = 380\n    FACE_MARGIN  = 0.30\n    NUM_FRAMES   = 32\n\n    EPOCHS       = 15\n    BATCH_SIZE   = 8\n    LR           = 1e-4\n    MIN_LR       = 1e-6\n    WEIGHT_DECAY = 1e-5\n    GRAD_CLIP    = 3.0\n    NUM_WORKERS  = 2\n    MIXED_PREC   = True\n    SEEDS        = [42, 1337, 777, 2021, 999]\n    CONFIDENT_T  = 0.8\n\n\nfor d in [CFG.CROPS_DIR, CFG.WEIGHTS_DIR]:\n    d.mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:21.585119Z","iopub.execute_input":"2026-03-31T14:15:21.585333Z","iopub.status.idle":"2026-03-31T14:15:21.591227Z","shell.execute_reply.started":"2026-03-31T14:15:21.585296Z","shell.execute_reply":"2026-03-31T14:15:21.590543Z"}},"outputs":[],"execution_count":null},{"id":"0d1366ff-26e0-40d1-a88e-80608264d345","cell_type":"markdown","source":"## 4. Face Detection & Crop Extraction","metadata":{}},{"id":"8c28388a-eadf-4e3d-b95e-9e10ad64804d","cell_type":"code","source":"face_detector = MTCNN(\n    margin=0,\n    thresholds=[0.6, 0.7, 0.7],\n    device=DEVICE,\n    keep_all=True,\n    post_process=False\n)\n\n\ndef get_video_scale(width, height):\n    max_side = max(width, height)\n    if max_side < 300:\n        return 2.0\n    elif max_side <= 1000:\n        return 1.0\n    elif max_side <= 1900:\n        return 0.5\n    return 0.33\n\n\ndef extract_face_crops(video_path, num_frames=CFG.NUM_FRAMES,\n                       margin_pct=CFG.FACE_MARGIN, target_size=CFG.INPUT_SIZE):\n    cap     = cv2.VideoCapture(video_path)\n    total   = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    width   = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\n    height  = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n    scale   = get_video_scale(width, height)\n    indices = np.linspace(0, total - 1, num_frames, dtype=int)\n    crops   = []\n\n    for idx in indices:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n        ret, frame = cap.read()\n        if not ret:\n            continue\n\n        if scale != 1.0:\n            det_frame = cv2.resize(frame, (int(width * scale), int(height * scale)))\n        else:\n            det_frame = frame\n\n        rgb = cv2.cvtColor(det_frame, cv2.COLOR_BGR2RGB)\n\n        with torch.no_grad():\n            boxes, _ = face_detector.detect(rgb)\n\n        if boxes is None or len(boxes) == 0:\n            h, w = frame.shape[:2]\n            s    = min(h, w)\n            y0   = (h - s) // 2\n            x0   = (w - s) // 2\n            crop = frame[y0:y0+s, x0:x0+s]\n            crops.append(cv2.resize(crop, (target_size, target_size)))\n            continue\n\n        boxes = boxes / scale\n        areas = [(b[2]-b[0])*(b[3]-b[1]) for b in boxes]\n        box   = boxes[np.argmax(areas)]\n        x1, y1, x2, y2 = [int(v) for v in box]\n\n        bw  = x2 - x1\n        bh  = y2 - y1\n        mx  = int(bw * margin_pct)\n        my  = int(bh * margin_pct)\n        x1m = max(0, x1 - mx)\n        y1m = max(0, y1 - my)\n        x2m = min(width,  x2 + mx)\n        y2m = min(height, y2 + my)\n\n        crop = frame[y1m:y2m, x1m:x2m]\n        if crop.size == 0:\n            crops.append(cv2.resize(frame, (target_size, target_size)))\n        else:\n            crops.append(cv2.resize(crop, (target_size, target_size)))\n\n    cap.release()\n    return crops","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:21.592538Z","iopub.execute_input":"2026-03-31T14:15:21.592770Z","iopub.status.idle":"2026-03-31T14:15:24.502287Z","shell.execute_reply.started":"2026-03-31T14:15:21.592724Z","shell.execute_reply":"2026-03-31T14:15:24.501603Z"}},"outputs":[],"execution_count":null},{"id":"fee220b4-4e5d-449e-8fbf-051c6314bd01","cell_type":"markdown","source":"## 5. Augmentation Pipeline","metadata":{}},{"id":"a5b60c89-647e-45c1-bf8a-46545096347a","cell_type":"code","source":"class IsotropicResize(A.DualTransform):\n    def __init__(self, max_side, interpolation_down=cv2.INTER_AREA,\n                 interpolation_up=cv2.INTER_CUBIC, always_apply=False, p=1):\n        super().__init__(always_apply, p)\n        self.max_side           = max_side\n        self.interpolation_down = interpolation_down\n        self.interpolation_up   = interpolation_up\n\n    def apply(self, img, **params):\n        h, w   = img.shape[:2]\n        if max(h, w) == self.max_side:\n            return img\n        interp = self.interpolation_down if max(h, w) > self.max_side else self.interpolation_up\n        scale  = self.max_side / max(h, w)\n        return cv2.resize(img, (int(w * scale), int(h * scale)), interpolation=interp)\n\n    def get_transform_init_args_names(self):\n        return ('max_side', 'interpolation_down', 'interpolation_up')\n\n\ndef create_train_transforms(size=CFG.INPUT_SIZE):\n    return A.Compose([\n        A.ImageCompression(quality_lower=60, quality_upper=100, p=0.5),\n        A.GaussNoise(p=0.1),\n        A.GaussianBlur(blur_limit=3, p=0.05),\n        A.HorizontalFlip(p=0.5),\n        A.OneOf([\n            IsotropicResize(max_side=size, interpolation_down=cv2.INTER_AREA,\n                            interpolation_up=cv2.INTER_CUBIC),\n            IsotropicResize(max_side=size, interpolation_down=cv2.INTER_AREA,\n                            interpolation_up=cv2.INTER_LINEAR),\n            IsotropicResize(max_side=size, interpolation_down=cv2.INTER_LINEAR,\n                            interpolation_up=cv2.INTER_LINEAR),\n        ], p=1),\n        A.PadIfNeeded(min_height=size, min_width=size, border_mode=cv2.BORDER_CONSTANT),\n        A.OneOf([\n            A.RandomBrightnessContrast(p=1),\n            A.HueSaturationValue(p=1),\n        ], p=0.7),\n        A.ToGray(p=0.2),\n        A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.2, rotate_limit=10,\n                           border_mode=cv2.BORDER_CONSTANT, p=0.5),\n        A.CoarseDropout(max_holes=8, max_height=size//8, max_width=size//8,\n                        min_holes=1, fill_value=0, p=0.3),\n        A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n        ToTensorV2()\n    ])\n\n\ndef create_val_transforms(size=CFG.INPUT_SIZE):\n    return A.Compose([\n        A.Resize(size, size),\n        A.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n        ToTensorV2()\n    ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:24.505747Z","iopub.execute_input":"2026-03-31T14:15:24.506081Z","iopub.status.idle":"2026-03-31T14:15:24.519825Z","shell.execute_reply.started":"2026-03-31T14:15:24.506019Z","shell.execute_reply":"2026-03-31T14:15:24.518994Z"}},"outputs":[],"execution_count":null},{"id":"ef7e9509-b7d2-41c9-8999-c8fd9067aac3","cell_type":"markdown","source":"## 6. Dataset","metadata":{}},{"id":"3d0553f2-0282-4553-a31e-162d7852cf38","cell_type":"code","source":"class DFDCDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df        = df.reset_index(drop=True)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row   = self.df.iloc[idx]\n        img   = cv2.imread(row['path'])\n        label = float(row['label'])\n\n        if img is None:\n            img = np.zeros((CFG.INPUT_SIZE, CFG.INPUT_SIZE, 3), dtype=np.uint8)\n        else:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        if self.transform:\n            img = self.transform(image=img)['image']\n\n        return img, torch.tensor(label, dtype=torch.float32)\n\n\ndef build_frame_index(video_dir, crops_dir, num_frames=CFG.NUM_FRAMES, max_videos=None):\n    with open(video_dir / 'metadata.json') as f:\n        metadata = json.load(f)\n\n    videos = list(video_dir.glob('*.mp4'))\n    if max_videos:\n        videos = videos[:max_videos]\n\n    records = []\n    print(f\"Processing {len(videos)} videos ...\")\n\n    for vpath in tqdm(videos):\n        vname   = vpath.name\n        label   = 1 if metadata.get(vname, {}).get('label', 'REAL') == 'FAKE' else 0\n        out_dir = crops_dir / vpath.stem\n        out_dir.mkdir(parents=True, exist_ok=True)\n\n        existing = list(out_dir.glob('*.jpg'))\n        if len(existing) >= num_frames:\n            for ep in existing:\n                records.append({'path': str(ep), 'label': label, 'video': vname})\n            continue\n\n        crops = extract_face_crops(str(vpath), num_frames=num_frames)\n        for i, crop in enumerate(crops):\n            cpath = out_dir / f'frame_{i:04d}.jpg'\n            cv2.imwrite(str(cpath), crop)\n            records.append({'path': str(cpath), 'label': label, 'video': vname})\n\n    return pd.DataFrame(records)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:24.521328Z","iopub.execute_input":"2026-03-31T14:15:24.521630Z","iopub.status.idle":"2026-03-31T14:15:24.549674Z","shell.execute_reply.started":"2026-03-31T14:15:24.521572Z","shell.execute_reply":"2026-03-31T14:15:24.548852Z"}},"outputs":[],"execution_count":null},{"id":"ea0ea256-b040-42e4-b254-82c1b2d425ca","cell_type":"markdown","source":"## 7. Model — EfficientNet-B7 Noisy Student","metadata":{}},{"id":"a2d3f0f3-107b-49d3-b414-c12ba44f4091","cell_type":"code","source":"class DeepFakeDetector(nn.Module):\n    def __init__(self, encoder=CFG.ENCODER, pretrained=True):\n        super().__init__()\n        self.backbone = timm.create_model(\n            encoder,\n            pretrained=pretrained,\n            num_classes=0,\n            global_pool='avg'\n        )\n        num_features = self.backbone.num_features\n        self.head = nn.Sequential(\n            nn.Linear(num_features, 512),\n            nn.SiLU(),\n            nn.Dropout(0.3),\n            nn.Linear(512, 1)\n        )\n        for m in self.head.modules():\n            if isinstance(m, nn.Linear):\n                nn.init.xavier_uniform_(m.weight)\n                nn.init.zeros_(m.bias)\n\n    def forward(self, x):\n        return self.head(self.backbone(x)).squeeze(1)\n\n\nmodel     = DeepFakeDetector(pretrained=True).to(DEVICE)\ntotal     = sum(p.numel() for p in model.parameters())\ntrainable = sum(p.numel() for p in model.parameters() if p.requires_grad)\nprint(f\"Total params     : {total/1e6:.1f}M\")\nprint(f\"Trainable params : {trainable/1e6:.1f}M\")\n\nx_dummy = torch.randn(2, 3, CFG.INPUT_SIZE, CFG.INPUT_SIZE).to(DEVICE)\nwith torch.no_grad():\n    out = model(x_dummy)\nprint(f\"Output shape     : {out.shape}\")\ndel model, x_dummy, out\ntorch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:24.550958Z","iopub.execute_input":"2026-03-31T14:15:24.551221Z","iopub.status.idle":"2026-03-31T14:15:28.410749Z","shell.execute_reply.started":"2026-03-31T14:15:24.551162Z","shell.execute_reply":"2026-03-31T14:15:28.409880Z"}},"outputs":[],"execution_count":null},{"id":"3c16da67-42a7-4b7b-8729-9664925d0b45","cell_type":"markdown","source":"## 8. Loss Function & Training Loop","metadata":{}},{"id":"96d431f1-b614-4a97-a311-7b43686efbd0","cell_type":"code","source":"class WeightedBCELoss(nn.Module):\n    def __init__(self, real_weight=1.0, fake_weight=1.5):\n        super().__init__()\n        self.real_weight = real_weight\n        self.fake_weight = fake_weight\n\n    def forward(self, logits, targets):\n        real_mask = targets == 0\n        fake_mask = targets == 1\n        loss      = torch.zeros(logits.shape[0], device=logits.device)\n        if real_mask.any():\n            loss[real_mask] = (\n                F.binary_cross_entropy_with_logits(\n                    logits[real_mask], targets[real_mask], reduction='none'\n                ) * self.real_weight\n            )\n        if fake_mask.any():\n            loss[fake_mask] = (\n                F.binary_cross_entropy_with_logits(\n                    logits[fake_mask], targets[fake_mask], reduction='none'\n                ) * self.fake_weight\n            )\n        return loss.mean()\n\n\ndef train_epoch(model, loader, optimizer, criterion, scaler):\n    model.train()\n    running_loss = 0.0\n    all_preds, all_labels = [], []\n\n    for imgs, labels in tqdm(loader, desc='Train', leave=False):\n        imgs   = imgs.to(DEVICE)\n        labels = labels.to(DEVICE)\n        optimizer.zero_grad()\n\n        if CFG.MIXED_PREC:\n            with autocast():\n                logits = model(imgs)\n                loss   = criterion(logits, labels)\n            scaler.scale(loss).backward()\n            scaler.unscale_(optimizer)\n            nn.utils.clip_grad_norm_(model.parameters(), CFG.GRAD_CLIP)\n            scaler.step(optimizer)\n            scaler.update()\n        else:\n            logits = model(imgs)\n            loss   = criterion(logits, labels)\n            loss.backward()\n            nn.utils.clip_grad_norm_(model.parameters(), CFG.GRAD_CLIP)\n            optimizer.step()\n\n        running_loss += loss.item()\n        all_preds.extend(torch.sigmoid(logits).detach().cpu().numpy())\n        all_labels.extend(labels.cpu().numpy())\n\n    avg_loss = running_loss / len(loader)\n    auc      = roc_auc_score(all_labels, all_preds) if len(set(all_labels)) > 1 else 0.5\n    return avg_loss, auc\n\n\n@torch.no_grad()\ndef val_epoch(model, loader, criterion):\n    model.eval()\n    running_loss = 0.0\n    all_preds, all_labels = [], []\n\n    for imgs, labels in tqdm(loader, desc='Val', leave=False):\n        imgs   = imgs.to(DEVICE)\n        labels = labels.to(DEVICE)\n        with autocast(enabled=CFG.MIXED_PREC):\n            logits = model(imgs)\n            loss   = criterion(logits, labels)\n        running_loss += loss.item()\n        all_preds.extend(torch.sigmoid(logits).cpu().numpy())\n        all_labels.extend(labels.cpu().numpy())\n\n    avg_loss = running_loss / len(loader)\n    logloss  = log_loss(all_labels, all_preds, eps=1e-7)\n    auc      = roc_auc_score(all_labels, all_preds) if len(set(all_labels)) > 1 else 0.5\n    return avg_loss, logloss, auc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:28.412320Z","iopub.execute_input":"2026-03-31T14:15:28.412629Z","iopub.status.idle":"2026-03-31T14:15:28.429433Z","shell.execute_reply.started":"2026-03-31T14:15:28.412568Z","shell.execute_reply":"2026-03-31T14:15:28.428563Z"}},"outputs":[],"execution_count":null},{"id":"101530d2-da10-41c7-8587-28070f8b884f","cell_type":"markdown","source":"## 9. Training Pipeline","metadata":{}},{"id":"1b2d00c6-ffbf-4665-b590-cb6fcd41c1c8","cell_type":"code","source":"def train_model(train_df, val_df, seed, model_idx):\n    seed_everything(seed)\n    print(f\"\\n{'='*55}\")\n    print(f\"  Model {model_idx+1}/{len(CFG.SEEDS)}   seed = {seed}\")\n    print(f\"{'='*55}\")\n\n    train_ds = DFDCDataset(train_df, transform=create_train_transforms())\n    val_ds   = DFDCDataset(val_df,   transform=create_val_transforms())\n\n    train_loader = DataLoader(train_ds, batch_size=CFG.BATCH_SIZE, shuffle=True,\n                              num_workers=CFG.NUM_WORKERS, pin_memory=True, drop_last=True)\n    val_loader   = DataLoader(val_ds,   batch_size=CFG.BATCH_SIZE * 2, shuffle=False,\n                              num_workers=CFG.NUM_WORKERS, pin_memory=True)\n\n    model     = DeepFakeDetector(pretrained=True).to(DEVICE)\n    criterion = WeightedBCELoss(real_weight=1.0, fake_weight=1.5)\n    optimizer = torch.optim.AdamW(model.parameters(), lr=CFG.LR, weight_decay=CFG.WEIGHT_DECAY)\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=CFG.EPOCHS, eta_min=CFG.MIN_LR)\n    scaler    = GradScaler(enabled=CFG.MIXED_PREC)\n\n    best_logloss = float('inf')\n    best_ckpt    = CFG.WEIGHTS_DIR / f'model_{model_idx}_seed{seed}.pth'\n    history      = {'train_loss': [], 'val_loss': [], 'val_logloss': [], 'val_auc': []}\n\n    for epoch in range(1, CFG.EPOCHS + 1):\n        lr = scheduler.get_last_lr()[0]\n        print(f\"\\n  Epoch {epoch}/{CFG.EPOCHS}   lr={lr:.2e}\")\n\n        t_loss, t_auc            = train_epoch(model, train_loader, optimizer, criterion, scaler)\n        v_loss, v_logloss, v_auc = val_epoch(model, val_loader, criterion)\n        scheduler.step()\n\n        history['train_loss'].append(t_loss)\n        history['val_loss'].append(v_loss)\n        history['val_logloss'].append(v_logloss)\n        history['val_auc'].append(v_auc)\n\n        print(f\"  Train  loss={t_loss:.4f}  auc={t_auc:.4f}\")\n        print(f\"  Val    loss={v_loss:.4f}  logloss={v_logloss:.4f}  auc={v_auc:.4f}\")\n\n        if v_logloss < best_logloss:\n            best_logloss = v_logloss\n            torch.save({'model_state': model.state_dict(),\n                        'epoch': epoch,\n                        'val_logloss': v_logloss,\n                        'val_auc': v_auc}, best_ckpt)\n            print(f\"  Checkpoint saved  (logloss={v_logloss:.4f})\")\n\n    return str(best_ckpt), history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:28.430802Z","iopub.execute_input":"2026-03-31T14:15:28.431134Z","iopub.status.idle":"2026-03-31T14:15:28.446640Z","shell.execute_reply.started":"2026-03-31T14:15:28.431077Z","shell.execute_reply":"2026-03-31T14:15:28.445957Z"}},"outputs":[],"execution_count":null},{"id":"c0d0e417-f364-4a3b-abb3-5299af2ec351","cell_type":"code","source":"# Run this to see exact contents of the dataset folder\nfrom pathlib import Path\n\nTRAIN_DIR = Path('/kaggle/input/competitions/deepfake-detection-challenge/train_sample_videos')\n\nprint(\"=== Files in TRAIN_DIR ===\")\nall_files = list(TRAIN_DIR.iterdir())\nfor f in sorted(all_files)[:30]:  # show first 30\n    print(f.name)\n\nprint(f\"\\nTotal files: {len(all_files)}\")\nprint(f\"\\n=== Video files (.mp4) ===\")\nmp4s = list(TRAIN_DIR.glob('*.mp4'))\nprint(f\"Count: {len(mp4s)}\")\nif mp4s:\n    print(\"Sample:\", mp4s[0].name)\n\nprint(f\"\\n=== JSON files ===\")\njsons = list(TRAIN_DIR.glob('*.json'))\nprint(f\"Count: {len(jsons)}\")\nif jsons:\n    import json\n    with open(jsons[0]) as f:\n        data = json.load(f)\n    print(\"Sample keys:\", list(data.keys())[:5])\n    first_key = list(data.keys())[0]\n    print(\"First entry:\", first_key, \"->\", data[first_key])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:28.447813Z","iopub.execute_input":"2026-03-31T14:15:28.448139Z","iopub.status.idle":"2026-03-31T14:15:28.500466Z","shell.execute_reply.started":"2026-03-31T14:15:28.448083Z","shell.execute_reply":"2026-03-31T14:15:28.499540Z"}},"outputs":[],"execution_count":null},{"id":"43f78fb2-c122-41d5-bf32-9ca972d2c468","cell_type":"markdown","source":"## 10. Build Frame Index","metadata":{}},{"id":"b2d64322-d81f-4190-8d44-40a177658eee","cell_type":"code","source":"def build_frame_index(video_dir, crops_dir, num_frames=10, max_videos=None):\n    import json, cv2\n    import pandas as pd\n    from pathlib import Path\n\n    video_dir  = Path(video_dir)\n    crops_dir  = Path(crops_dir)\n    crops_dir.mkdir(parents=True, exist_ok=True)\n\n    # Load metadata JSON\n    json_files = list(video_dir.glob('*.json'))\n    assert len(json_files) == 1, f\"Expected 1 JSON, found {len(json_files)}\"\n    with open(json_files[0]) as f:\n        meta = json.load(f)\n\n    # Build records\n    records = []\n    videos  = list(video_dir.glob('*.mp4'))\n    if max_videos:\n        videos = videos[:max_videos]\n\n    for video_path in tqdm(videos, desc=\"Indexing frames\"):\n        fname = video_path.name\n        if fname not in meta:\n            continue\n        label = 1 if meta[fname]['label'] == 'FAKE' else 0\n\n        cap = cv2.VideoCapture(str(video_path))\n        total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        if total <= 0:\n            cap.release()\n            continue\n\n        indices = np.linspace(0, total - 1, num_frames, dtype=int)\n        for idx in indices:\n            cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n            ret, frame = cap.read()\n            if not ret:\n                continue\n            frame_name = f\"{fname[:-4]}_f{idx:04d}.jpg\"\n            frame_path = crops_dir / frame_name\n            if not frame_path.exists():\n                cv2.imwrite(str(frame_path), frame)\n            records.append({\n                'video' : fname,\n                'frame' : str(frame_path),\n                'label' : label,\n                'original': meta[fname].get('original', '')\n            })\n        cap.release()\n\n    df = pd.DataFrame(records)\n    if len(df) == 0:\n        raise ValueError(\"❌ No data found — check dataset path!\")\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:28.501613Z","iopub.execute_input":"2026-03-31T14:15:28.501925Z","iopub.status.idle":"2026-03-31T14:15:28.513210Z","shell.execute_reply.started":"2026-03-31T14:15:28.501870Z","shell.execute_reply":"2026-03-31T14:15:28.512321Z"}},"outputs":[],"execution_count":null},{"id":"feed8912-6acf-4343-ab81-d75f9c157da2","cell_type":"markdown","source":"## 11. Train / Validation Split","metadata":{}},{"id":"3fca64c8-5626-47a1-8c6d-6069025785a1","cell_type":"code","source":"INDEX_PATH = Path('/kaggle/working/frame_index.csv')\n\nif INDEX_PATH.exists():\n    frame_df = pd.read_csv(INDEX_PATH)\n    print(\"Frame index loaded from cache.\")\nelse:\n    TRAIN_DIR = Path('/kaggle/input/competitions/deepfake-detection-challenge/train_sample_videos')\n    CROPS_DIR = Path('/kaggle/working/crops')\n    CROPS_DIR.mkdir(parents=True, exist_ok=True)\n\n    frame_df = build_frame_index(TRAIN_DIR, CROPS_DIR)\n    frame_df.to_csv(INDEX_PATH, index=False)\n\nprint(f\"Total frames  : {len(frame_df):,}\")\nprint(f\"Total videos  : {frame_df['video'].nunique():,}\")\nprint(f\"Fake frames   : {frame_df['label'].sum():,.0f}  ({frame_df['label'].mean()*100:.1f}%)\")\nprint(f\"Real frames   : {(frame_df['label']==0).sum():,}\")\nframe_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:15:28.514427Z","iopub.execute_input":"2026-03-31T14:15:28.514616Z","iopub.status.idle":"2026-03-31T14:47:01.554316Z","shell.execute_reply.started":"2026-03-31T14:15:28.514582Z","shell.execute_reply":"2026-03-31T14:47:01.553569Z"}},"outputs":[],"execution_count":null},{"id":"6a4be0aa-3454-4196-99d0-c0c4b455cb2c","cell_type":"markdown","source":"## 12. Run Training","metadata":{}},{"id":"95668654-107e-4685-82ad-8ada6ba68820","cell_type":"code","source":"from sklearn.model_selection import GroupShuffleSplit\nfrom torch.cuda.amp import autocast, GradScaler\n\n\nif 'CFG' not in globals() and 'cfg' in globals():\n    CFG = cfg\n\nif 'CFG' not in globals():\n    raise NameError(\"CFG is not defined. Run section 3 first.\")\n\nseed_list = getattr(CFG, 'SEEDS', None)\n\nif not seed_list and 'cfg' in globals():\n    seed_list = getattr(cfg, 'seeds', None)\n\nif not seed_list and 'seeds' in globals():\n    seed_list = seeds\n\nif not seed_list:\n    seed_list = [globals().get('SEED', 42)]\n\nif 'train_df' not in globals() or 'val_df' not in globals():\n    if 'frame_df' not in globals():\n        raise NameError(\"frame_df is not defined. Run section 11 first.\")\n\n    if 'frame' in frame_df.columns and 'path' not in frame_df.columns:\n        frame_df = frame_df.rename(columns={'frame': 'path'})\n\n    if 'video' not in frame_df.columns:\n        raise ValueError(\"frame_df must contain a 'video' column.\")\n\n    required_cols = {'path', 'label', 'video'}\n    missing_cols = required_cols - set(frame_df.columns)\n    if missing_cols:\n        raise ValueError(f\"frame_df is missing required columns: {missing_cols}\")\n\n    frame_df = frame_df.dropna(subset=['path', 'label', 'video']).copy()\n    frame_df['label'] = frame_df['label'].astype(int)\n\n    split_seed = globals().get('SEED', seed_list[0] if len(seed_list) else 42)\n    splitter = GroupShuffleSplit(n_splits=1, test_size=0.20, random_state=split_seed)\n\n    train_idx, val_idx = next(\n        splitter.split(frame_df, y=frame_df['label'], groups=frame_df['video'])\n    )\n\n    train_df = frame_df.iloc[train_idx].reset_index(drop=True)\n    val_df = frame_df.iloc[val_idx].reset_index(drop=True)\n\n    print(f\"Created split from frame_df: train={len(train_df):,}, val={len(val_df):,}\")\n\nall_checkpoints = []\nall_histories = []\n\nfor i, seed in enumerate(seed_list):\n    ckpt, hist = train_model(train_df, val_df, seed=seed, model_idx=i)\n    all_checkpoints.append(ckpt)\n    all_histories.append(hist)\n\nprint(f\"\\nTraining complete - {len(all_checkpoints)} models\")\nfor i, ckpt in enumerate(all_checkpoints):\n    m = torch.load(ckpt, map_location='cpu')\n    print(f\"Model {i+1}  epoch={m['epoch']}  logloss={m['val_logloss']:.4f}  auc={m['val_auc']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T14:47:34.396606Z","iopub.execute_input":"2026-03-31T14:47:34.396899Z","iopub.status.idle":"2026-03-31T20:16:29.930448Z","shell.execute_reply.started":"2026-03-31T14:47:34.396859Z","shell.execute_reply":"2026-03-31T20:16:29.929541Z"}},"outputs":[],"execution_count":null},{"id":"fe8b272f-2cad-47c9-b611-f67a3d0cacc2","cell_type":"code","source":"colors = ['#e74c3c', '#3498db', '#2ecc71', '#f39c12', '#9b59b6']\nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\nfor i, (hist, seed) in enumerate(zip(all_histories, CFG.SEEDS)):\n    ep = range(1, len(hist['val_logloss']) + 1)\n    axes[0].plot(ep, hist['train_loss'], color=colors[i], linestyle='--', alpha=0.5, label=f'Train s={seed}')\n    axes[0].plot(ep, hist['val_loss'],   color=colors[i], linestyle='-',  label=f'Val s={seed}')\n    axes[1].plot(ep, hist['val_logloss'], color=colors[i], label=f'seed={seed}')\n    axes[2].plot(ep, hist['val_auc'],     color=colors[i], label=f'seed={seed}')\n\nfor ax, title, ylabel in zip(axes,\n                              ['BCE Loss', 'Val Log-Loss', 'Val AUC'],\n                              ['Loss', 'Log-Loss', 'AUC']):\n    ax.set_title(title, fontsize=13, fontweight='bold')\n    ax.set_xlabel('Epoch')\n    ax.set_ylabel(ylabel)\n    ax.legend(fontsize=8)\n    ax.grid(True, alpha=0.3)\n\nplt.suptitle('DeepFake Detection — Training Curves', fontsize=15, fontweight='bold', y=1.02)\nplt.tight_layout()\nplt.savefig('/kaggle/working/training_curves.png', dpi=150, bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T20:31:26.349471Z","iopub.execute_input":"2026-03-31T20:31:26.349793Z","iopub.status.idle":"2026-03-31T20:31:28.766406Z","shell.execute_reply.started":"2026-03-31T20:31:26.349730Z","shell.execute_reply":"2026-03-31T20:31:28.765580Z"}},"outputs":[],"execution_count":null},{"id":"225732f3-e63d-4df6-aae8-cdaf762605fa","cell_type":"markdown","source":"## 13. Training Curves","metadata":{}},{"id":"dd47e0bc-8593-4c67-8fdb-99c4020cf22a","cell_type":"markdown","source":"## 14. Confident Strategy Aggregation","metadata":{}},{"id":"5b31ac3f-c615-4d5e-8e9c-f0d39674c88b","cell_type":"code","source":"def confident_strategy(pred, t=CFG.CONFIDENT_T):\n    pred  = np.array(pred)\n    sz    = len(pred)\n    fakes = np.count_nonzero(pred > t)\n    if fakes > sz // 2.5 and fakes > 11:\n        return np.mean(pred[pred > t])\n    elif np.count_nonzero(pred < 0.2) > 0.9 * sz:\n        return np.mean(pred[pred < 0.2])\n    else:\n        return np.mean(pred)\n\n\n@torch.no_grad()\ndef predict_video(video_path, models, num_frames=CFG.NUM_FRAMES):\n    crops = extract_face_crops(video_path, num_frames=num_frames)\n    if not crops:\n        return 0.5\n\n    val_tf = create_val_transforms()\n    batch  = torch.stack([\n        val_tf(image=cv2.cvtColor(c, cv2.COLOR_BGR2RGB))['image']\n        for c in crops\n    ]).to(DEVICE)\n\n    ensemble_preds = []\n    for model in models:\n        model.eval()\n        sub_bs = 8\n        logits = []\n        for i in range(0, len(batch), sub_bs):\n            sub = batch[i:i+sub_bs]\n            with autocast(enabled=CFG.MIXED_PREC):\n                logits.append(model(sub))\n        logits = torch.cat(logits)\n        probs  = torch.sigmoid(logits).cpu().numpy()\n        ensemble_preds.append(confident_strategy(probs))\n\n    return float(np.mean(ensemble_preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T20:32:43.147628Z","iopub.execute_input":"2026-03-31T20:32:43.147892Z","iopub.status.idle":"2026-03-31T20:32:43.157592Z","shell.execute_reply.started":"2026-03-31T20:32:43.147852Z","shell.execute_reply":"2026-03-31T20:32:43.156807Z"}},"outputs":[],"execution_count":null},{"id":"bd5ae2ea-3cde-4bd5-9404-42d8b1101747","cell_type":"markdown","source":"## 15. Load Ensemble","metadata":{}},{"id":"df749d62-581f-40e5-af6c-41f8dd47561c","cell_type":"code","source":"ensemble_models = []\n\nfor ckpt_path in all_checkpoints:\n    m     = DeepFakeDetector(pretrained=False).to(DEVICE)\n    state = torch.load(ckpt_path, map_location=DEVICE)\n    m.load_state_dict(state['model_state'])\n    m.eval()\n    ensemble_models.append(m)\n\nprint(f\"Ensemble ready — {len(ensemble_models)} models loaded\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T20:32:47.543049Z","iopub.execute_input":"2026-03-31T20:32:47.543336Z","iopub.status.idle":"2026-03-31T20:32:54.566935Z","shell.execute_reply.started":"2026-03-31T20:32:47.543282Z","shell.execute_reply":"2026-03-31T20:32:54.566065Z"}},"outputs":[],"execution_count":null},{"id":"8639323e-dbec-48d1-8ed5-e165f281d9ec","cell_type":"markdown","source":"## 16. Test Set Inference","metadata":{}},{"id":"3ab273d9-1072-4957-ba9d-55223f351f42","cell_type":"code","source":"@torch.no_grad()\ndef extract_test_frames(video_path, num_frames=CFG.NUM_FRAMES):\n    cap = cv2.VideoCapture(str(video_path))\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n\n    if total <= 0:\n        cap.release()\n        return []\n\n    indices = np.linspace(0, total - 1, num_frames, dtype=int)\n    frames = []\n\n    for idx in indices:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, int(idx))\n        ret, frame = cap.read()\n        if not ret or frame is None:\n            continue\n\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        frames.append(frame)\n\n    cap.release()\n    return frames\n\n\n@torch.no_grad()\ndef predict_video(video_path, models, num_frames=CFG.NUM_FRAMES):\n    frames = extract_test_frames(video_path, num_frames=num_frames)\n\n    if len(frames) == 0:\n        return 0.5\n\n    val_tf = create_val_transforms()\n    batch = torch.stack([\n        val_tf(image=frame)['image'] for frame in frames\n    ]).to(DEVICE)\n\n    ensemble_preds = []\n\n    for model in models:\n        model.eval()\n        logits = []\n        sub_bs = 8\n\n        for i in range(0, len(batch), sub_bs):\n            sub = batch[i:i + sub_bs]\n            logits.append(model(sub))\n\n        logits = torch.cat(logits)\n        probs = torch.sigmoid(logits).detach().cpu().numpy()\n\n        if 'confident_strategy' in globals():\n            ensemble_preds.append(confident_strategy(probs))\n        else:\n            ensemble_preds.append(float(np.mean(probs)))\n\n    return float(np.mean(ensemble_preds))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T20:45:44.576524Z","iopub.execute_input":"2026-03-31T20:45:44.576852Z","iopub.status.idle":"2026-03-31T20:45:44.587709Z","shell.execute_reply.started":"2026-03-31T20:45:44.576795Z","shell.execute_reply":"2026-03-31T20:45:44.586969Z"}},"outputs":[],"execution_count":null},{"id":"a9581d57-c7e7-4bf9-ab10-dfa3cf95a794","cell_type":"markdown","source":"## 17. Validation Metrics","metadata":{}},{"id":"da6f0f4b-98ad-4fc2-a54c-1fcd617a349e","cell_type":"code","source":"print(\"CFG.TRAIN_DIR =\", CFG.TRAIN_DIR)\nprint(val_df[['video']].head())\n\nif 'video_path' in val_df.columns:\n    print(val_df[['video_path']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T20:50:39.599548Z","iopub.execute_input":"2026-03-31T20:50:39.599866Z","iopub.status.idle":"2026-03-31T20:50:39.607264Z","shell.execute_reply.started":"2026-03-31T20:50:39.599807Z","shell.execute_reply":"2026-03-31T20:50:39.606535Z"}},"outputs":[],"execution_count":null},{"id":"ea3ed237-6dc3-47a9-9921-2d1e63c8c28b","cell_type":"code","source":"from pathlib import Path\n\nif 'val_df' not in globals():\n    raise NameError(\"val_df is not defined. Run the split cell first.\")\n\n# Auto-detect the actual train_sample_videos folder in this Kaggle session\ntrain_candidates = sorted(Path('/kaggle/input').rglob('train_sample_videos'))\n\nif not train_candidates:\n    raise FileNotFoundError(\"Could not find any train_sample_videos folder under /kaggle/input\")\n\ntrain_dir = train_candidates[0]\nCFG.TRAIN_DIR = train_dir\n\nprint(\"Using TRAIN_DIR:\", train_dir)\n\nval_video_df = (\n    val_df[['video', 'label']]\n    .drop_duplicates(subset='video')\n    .reset_index(drop=True)\n)\n\neval_video_df = val_video_df.head(100).copy()\n\nvideo_preds = []\nvideo_labels_gt = []\nmissing_paths = []\n\nfor row in tqdm(eval_video_df.itertuples(index=False), total=len(eval_video_df), desc='Val inference'):\n    vpath = train_dir / row.video\n\n    if not vpath.exists():\n        missing_paths.append(str(vpath))\n        continue\n\n    prob = predict_video(str(vpath), ensemble_models)\n    video_preds.append(float(prob))\n    video_labels_gt.append(int(row.label))\n\nif missing_paths:\n    print(\"Missing video files:\", len(missing_paths))\n    print(\"Sample missing paths:\")\n    for p in missing_paths[:5]:\n        print(\" -\", p)\n\nif len(video_labels_gt) == 0:\n    raise ValueError(\n        \"No validation videos were found at the detected train directory. \"\n        \"If this still happens, rebuild frame_df/train_df/val_df from the current Kaggle input.\"\n    )\n\npred_labels = (np.array(video_preds) > 0.5).astype(int)\nacc = (pred_labels == np.array(video_labels_gt)).mean()\n\nprint(f\"{'='*45}\")\nprint(f\"  Accuracy : {acc*100:.2f}%\")\nprint(f\"  Videos   : {len(video_labels_gt)}\")\nprint(f\"  Fake     : {sum(video_labels_gt)}\")\nprint(f\"  Real     : {len(video_labels_gt) - sum(video_labels_gt)}\")\n\nif len(set(video_labels_gt)) > 1:\n    ll = log_loss(video_labels_gt, video_preds, eps=1e-7)\n    auc = roc_auc_score(video_labels_gt, video_preds)\n    print(f\"  Log-Loss : {ll:.4f}\")\n    print(f\"  ROC-AUC  : {auc:.4f}\")\n\nprint(f\"{'='*45}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T20:52:15.349542Z","iopub.execute_input":"2026-03-31T20:52:15.349800Z","iopub.status.idle":"2026-03-31T21:15:03.145314Z","shell.execute_reply.started":"2026-03-31T20:52:15.349762Z","shell.execute_reply":"2026-03-31T21:15:03.144567Z"}},"outputs":[],"execution_count":null},{"id":"f621c265-6558-477e-b08e-94d4dd467167","cell_type":"code","source":"face_cascade = cv2.CascadeClassifier(\n    cv2.data.haarcascades + 'haarcascade_frontalface_default.xml'\n)\n\n@torch.no_grad()\ndef extract_face_crops_cv(video_path, num_frames=CFG.NUM_FRAMES, margin_pct=CFG.FACE_MARGIN):\n    cap = cv2.VideoCapture(str(video_path))\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n\n    if total <= 0:\n        cap.release()\n        return []\n\n    indices = np.linspace(0, total - 1, num_frames, dtype=int)\n    crops = []\n\n    for idx in indices:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, int(idx))\n        ret, frame = cap.read()\n\n        if not ret or frame is None:\n            continue\n\n        gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n        faces = face_cascade.detectMultiScale(\n            gray,\n            scaleFactor=1.1,\n            minNeighbors=5,\n            minSize=(60, 60)\n        )\n\n        if len(faces) > 0:\n            x, y, w, h = max(faces, key=lambda b: b[2] * b[3])\n            mx = int(w * margin_pct)\n            my = int(h * margin_pct)\n\n            x1 = max(0, x - mx)\n            y1 = max(0, y - my)\n            x2 = min(frame.shape[1], x + w + mx)\n            y2 = min(frame.shape[0], y + h + my)\n            crop = frame[y1:y2, x1:x2]\n        else:\n            h0, w0 = frame.shape[:2]\n            side = min(h0, w0)\n            x1 = (w0 - side) // 2\n            y1 = (h0 - side) // 2\n            crop = frame[y1:y1 + side, x1:x1 + side]\n\n        if crop is None or crop.size == 0:\n            continue\n\n        crop = cv2.cvtColor(crop, cv2.COLOR_BGR2RGB)\n        crops.append(crop)\n\n    cap.release()\n    return crops\n\n\n@torch.no_grad()\ndef predict_video(video_path, models, num_frames=CFG.NUM_FRAMES):\n    crops = extract_face_crops_cv(video_path, num_frames=num_frames)\n\n    if len(crops) == 0:\n        return 0.5\n\n    val_tf = create_val_transforms()\n    batch = torch.stack([val_tf(image=crop)['image'] for crop in crops]).to(DEVICE)\n\n    ensemble_preds = []\n\n    for model in models:\n        model.eval()\n        logits = []\n\n        for i in range(0, len(batch), 8):\n            sub = batch[i:i + 8]\n            logits.append(model(sub))\n\n        logits = torch.cat(logits)\n        probs = torch.sigmoid(logits).detach().cpu().numpy()\n\n        if 'confident_strategy' in globals():\n            ensemble_preds.append(float(confident_strategy(probs)))\n        else:\n            ensemble_preds.append(float(np.mean(probs)))\n\n    return float(np.mean(ensemble_preds))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T21:19:14.790498Z","iopub.execute_input":"2026-03-31T21:19:14.790764Z","iopub.status.idle":"2026-03-31T21:19:14.829620Z","shell.execute_reply.started":"2026-03-31T21:19:14.790727Z","shell.execute_reply":"2026-03-31T21:19:14.829118Z"}},"outputs":[],"execution_count":null},{"id":"8a274423-e2bf-41c5-901d-699dd82fe561","cell_type":"code","source":"scores = np.array(video_preds)\ntargets = np.array(video_labels_gt)\n\nbest_t = 0.50\nbest_acc = 0.0\n\nfor t in np.arange(0.05, 0.96, 0.01):\n    preds = (scores >= t).astype(int)\n    acc = (preds == targets).mean()\n\n    if acc > best_acc:\n        best_acc = acc\n        best_t = t\n\nprint(f\"Best threshold: {best_t:.2f}\")\nprint(f\"Best accuracy : {best_acc*100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T21:19:36.595209Z","iopub.execute_input":"2026-03-31T21:19:36.595478Z","iopub.status.idle":"2026-03-31T21:19:36.603561Z","shell.execute_reply.started":"2026-03-31T21:19:36.595439Z","shell.execute_reply":"2026-03-31T21:19:36.602825Z"}},"outputs":[],"execution_count":null},{"id":"62c6f8f5-8596-40c0-867a-3c6d9d146418","cell_type":"markdown","source":"## 18. Results Visualisation","metadata":{}},{"id":"fca20f78-1d1a-44e7-9793-d253a2e680e0","cell_type":"code","source":"if video_labels_gt:\n    real_preds = [p for p, l in zip(video_preds, video_labels_gt) if l == 0]\n    fake_preds = [p for p, l in zip(video_preds, video_labels_gt) if l == 1]\n\n    fig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n    axes[0].hist(real_preds, bins=30, alpha=0.7, color='#2ecc71', label='Real', density=True)\n    axes[0].hist(fake_preds, bins=30, alpha=0.7, color='#e74c3c', label='Fake', density=True)\n    axes[0].axvline(0.5, color='black', linestyle='--', linewidth=1.5, label='Threshold = 0.5')\n    axes[0].set_title('Prediction Distribution', fontsize=13, fontweight='bold')\n    axes[0].set_xlabel('Fake Probability')\n    axes[0].set_ylabel('Density')\n    axes[0].legend()\n    axes[0].grid(True, alpha=0.3)\n\n    fpr, tpr, _ = roc_curve(video_labels_gt, video_preds)\n    axes[1].plot(fpr, tpr, color='#3498db', lw=2, label=f'AUC = {auc:.3f}')\n    axes[1].plot([0, 1], [0, 1], 'k--', lw=1)\n    axes[1].fill_between(fpr, tpr, alpha=0.1, color='#3498db')\n    axes[1].set_title('ROC Curve', fontsize=13, fontweight='bold')\n    axes[1].set_xlabel('False Positive Rate')\n    axes[1].set_ylabel('True Positive Rate')\n    axes[1].legend()\n    axes[1].grid(True, alpha=0.3)\n\n    plt.suptitle('DeepFake Detection — Validation Results', fontsize=15, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig('/kaggle/working/validation_results.png', dpi=150, bbox_inches='tight')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T21:15:03.147271Z","iopub.execute_input":"2026-03-31T21:15:03.147473Z","iopub.status.idle":"2026-03-31T21:15:04.344829Z","shell.execute_reply.started":"2026-03-31T21:15:03.147439Z","shell.execute_reply":"2026-03-31T21:15:04.343358Z"}},"outputs":[],"execution_count":null}]}