{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required libraries\n!pip install torchvision","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:48:40.475031Z","iopub.execute_input":"2025-03-04T14:48:40.475379Z","iopub.status.idle":"2025-03-04T14:48:45.141475Z","shell.execute_reply.started":"2025-03-04T14:48:40.475351Z","shell.execute_reply":"2025-03-04T14:48:45.139722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required libraries\n!pip install timm albumentations facenet-pytorch dlib wandb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:40:56.006315Z","iopub.execute_input":"2025-03-04T14:40:56.006957Z","iopub.status.idle":"2025-03-04T14:44:18.543139Z","shell.execute_reply.started":"2025-03-04T14:40:56.00685Z","shell.execute_reply":"2025-03-04T14:44:18.541774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install necessary packages\n!pip install -q facenet-pytorch mtcnn efficientnet_pytorch pytorch-lightning icarl gdumb\n!pip install -q albumentations onnx onnxruntime tensorrt\n\n# Import required libraries\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport cv2\nimport json\nfrom PIL import Image\nimport time\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# PyTorch imports\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nimport torchvision.transforms.functional as TF\n\n# Face detection & alignment\nfrom facenet_pytorch import MTCNN\n\n# EfficientNet\nfrom efficientnet_pytorch import EfficientNet\n\n# For metrics\nfrom sklearn.metrics import accuracy_score, roc_auc_score, confusion_matrix, average_precision_score\n\n# Set random seeds for reproducibility\nseed = 42\nrandom.seed(seed)\nnp.random.seed(seed)\ntorch.manual_seed(seed)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n\n# Check if GPU is available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T17:15:13.850375Z","iopub.execute_input":"2025-03-04T17:15:13.850865Z","iopub.status.idle":"2025-03-04T17:15:25.369772Z","shell.execute_reply.started":"2025-03-04T17:15:13.850784Z","shell.execute_reply":"2025-03-04T17:15:25.368829Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset Selection and Preparation","metadata":{}},{"cell_type":"code","source":"# Define the datasets to use - adjust paths based on available Kaggle datasets\nDATASET_DIR = \"/kaggle/input\"\n\n# You may need to adjust these paths depending on which datasets you've added to your notebook\nDATASETS = {\n    \"ff++\": f\"{DATASET_DIR}/faceforensics\",  # FaceForensics++\n    \"celebdf\": f\"{DATASET_DIR}/celeb-df\",    # Celeb-DF\n    \"dfdc\": f\"{DATASET_DIR}/deepfake-detection-challenge\",  # DFDC\n}\n\n# Function to list available datasets in your Kaggle environment\ndef list_available_datasets():\n    print(\"Available datasets:\")\n    for directory in os.listdir(DATASET_DIR):\n        print(f\" - {directory}\")\n    \nlist_available_datasets()\n\n# Create directories for processed data\nos.makedirs(\"/kaggle/working/processed_data\", exist_ok=True)\nos.makedirs(\"/kaggle/working/models\", exist_ok=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading Helper Functions","metadata":{}},{"cell_type":"code","source":"class DeepfakeDataset(Dataset):\n    def __init__(self, image_paths, labels, transform=None):\n        self.image_paths = image_paths\n        self.labels = labels\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, idx):\n        img_path = self.image_paths[idx]\n        image = Image.open(img_path).convert('RGB')\n        label = self.labels[idx]\n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label\n\ndef create_data_loaders(train_paths, train_labels, valid_paths, valid_labels, batch_size=32):\n    # Define transformations for training and validation\n    train_transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.RandomHorizontalFlip(),\n        transforms.ColorJitter(brightness=0.1, contrast=0.1, saturation=0.1),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    valid_transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    # Create datasets\n    train_dataset = DeepfakeDataset(train_paths, train_labels, train_transform)\n    valid_dataset = DeepfakeDataset(valid_paths, valid_labels, valid_transform)\n    \n    # Create data loaders\n    train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=2, pin_memory=True)\n    valid_loader = DataLoader(valid_dataset, batch_size=batch_size, shuffle=False, num_workers=2, pin_memory=True)\n    \n    return train_loader, valid_loader","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Face Detection and Processing","metadata":{}},{"cell_type":"code","source":"def extract_faces(image_path, mtcnn, save_path=None, min_face_size=50):\n    \"\"\"Extract faces from an image and optionally save them\"\"\"\n    try:\n        img = cv2.imread(image_path)\n        if img is None:\n            print(f\"Error loading image: {image_path}\")\n            return None\n            \n        # Convert to RGB for MTCNN\n        img_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Detect faces\n        boxes, probs = mtcnn.detect(img_rgb)\n        \n        # If no face detected\n        if boxes is None:\n            return None\n            \n        faces = []\n        \n        for i, (box, prob) in enumerate(zip(boxes, probs)):\n            if prob < 0.9:  # Filter by detection confidence\n                continue\n                \n            # Get coordinates\n            x1, y1, x2, y2 = [int(coord) for coord in box]\n            \n            # Skip if face is too small\n            if (x2 - x1 < min_face_size) or (y2 - y1 < min_face_size):\n                continue\n            \n            # Extract face with a margin\n            h, w = img.shape[:2]\n            margin = 20\n            x1 = max(0, x1 - margin)\n            y1 = max(0, y1 - margin)\n            x2 = min(w, x2 + margin)\n            y2 = min(h, y2 + margin)\n            \n            face = img_rgb[y1:y2, x1:x2]\n            \n            if face.size == 0:\n                continue\n                \n            # Convert back to BGR for saving\n            face_bgr = cv2.cvtColor(face, cv2.COLOR_RGB2BGR)\n            \n            # Save if path provided\n            if save_path:\n                face_filename = f\"{os.path.splitext(os.path.basename(image_path))[0]}_face_{i}.jpg\"\n                face_path = os.path.join(save_path, face_filename)\n                cv2.imwrite(face_path, face_bgr)\n                faces.append(face_path)\n            else:\n                faces.append(face)\n                \n        return faces\n    except Exception as e:\n        print(f\"Error processing {image_path}: {str(e)}\")\n        return None\n\ndef preprocess_dataset(dataset_dir, output_dir, is_fake=False):\n    \"\"\"Process a dataset directory, extract faces and prepare for training\"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Initialize MTCNN for face detection\n    mtcnn = MTCNN(keep_all=True, device=device)\n    \n    image_paths = []\n    for root, _, files in tqdm(os.walk(dataset_dir), desc=\"Finding images\"):\n        for file in files:\n            if file.lower().endswith(('.png', '.jpg', '.jpeg')):\n                image_paths.append(os.path.join(root, file))\n    \n    processed_paths = []\n    for img_path in tqdm(image_paths, desc=\"Processing images\"):\n        faces = extract_faces(img_path, mtcnn, output_dir)\n        if faces:\n            processed_paths.extend(faces)\n    \n    # Create labels (1 for fake, 0 for real)\n    labels = [1 if is_fake else 0] * len(processed_paths)\n    \n    return processed_paths, labels","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Spectral Analysis Features","metadata":{}},{"cell_type":"code","source":"def extract_spectral_features(image_path):\n    \"\"\"Extract spectral features using DFT\"\"\"\n    try:\n        # Read the image\n        img = cv2.imread(image_path)\n        if img is None:\n            return None\n            \n        # Convert to grayscale\n        gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        \n        # Apply DFT\n        dft = cv2.dft(np.float32(gray), flags=cv2.DFT_COMPLEX_OUTPUT)\n        dft_shift = np.fft.fftshift(dft)\n        \n        # Calculate magnitude spectrum\n        magnitude_spectrum = 20 * np.log(cv2.magnitude(dft_shift[:,:,0], dft_shift[:,:,1]) + 1)\n        \n        # Resize to a standard size\n        magnitude_spectrum_resized = cv2.resize(magnitude_spectrum, (128, 128))\n        \n        # Normalize\n        magnitude_spectrum_norm = (magnitude_spectrum_resized - np.min(magnitude_spectrum_resized)) / \\\n                                  (np.max(magnitude_spectrum_resized) - np.min(magnitude_spectrum_resized) + 1e-8)\n        \n        return magnitude_spectrum_norm\n    except Exception as e:\n        print(f\"Error processing spectrum for {image_path}: {str(e)}\")\n        return None\n\ndef add_spectral_channel(dataset_paths, output_dir):\n    \"\"\"Process dataset to add spectral features as an additional channel\"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    \n    enhanced_paths = []\n    \n    for img_path in tqdm(dataset_paths, desc=\"Extracting spectral features\"):\n        # Get base name without extension\n        base_name = os.path.splitext(os.path.basename(img_path))[0]\n        \n        # Get spectral features\n        spectral_features = extract_spectral_features(img_path)\n        if spectral_features is None:\n            continue\n            \n        # Read original image\n        img = cv2.imread(img_path)\n        \n        # Create a new image with 4 channels (RGB + Spectral)\n        # For simplicity, we'll save the spectral features as a separate grayscale image\n        spectral_path = os.path.join(output_dir, f\"{base_name}_spectral.png\")\n        cv2.imwrite(spectral_path, (spectral_features * 255).astype(np.uint8))\n        \n        # Keep the original image path for loading\n        enhanced_paths.append((img_path, spectral_path))\n        \n    return enhanced_paths","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Processing Pipeline","metadata":{}},{"cell_type":"code","source":"def process_datasets():\n    \"\"\"Run the complete data processing pipeline\"\"\"\n    # Initialize MTCNN\n    mtcnn = MTCNN(keep_all=True, device=device)\n    \n    # Process datasets based on what's available\n    all_processed_data = {\n        'train': {'real': [], 'fake': []},\n        'val': {'real': [], 'fake': []}\n    }\n    \n    # Sample code to process FaceForensics++\n    if os.path.exists(DATASETS.get('ff++', '')):\n        print(\"Processing FaceForensics++...\")\n        # Adjust these paths based on FF++ dataset structure\n        ff_real_dir = os.path.join(DATASETS['ff++'], 'original_sequences/youtube/c23/frames')\n        ff_fake_dir = os.path.join(DATASETS['ff++'], 'manipulated_sequences/Deepfakes/c23/frames')\n        \n        # Create output directories\n        real_output_dir = \"/kaggle/working/processed_data/ff_real\"\n        fake_output_dir = \"/kaggle/working/processed_data/ff_fake\"\n        os.makedirs(real_output_dir, exist_ok=True)\n        os.makedirs(fake_output_dir, exist_ok=True)\n        \n        # Get list of directories (videos)\n        real_videos = os.listdir(ff_real_dir)\n        fake_videos = os.listdir(ff_fake_dir)\n        \n        # Split for train/val\n        train_ratio = 0.8\n        \n        # Process real videos\n        train_count = int(len(real_videos) * train_ratio)\n        train_real_videos = real_videos[:train_count]\n        val_real_videos = real_videos[train_count:]\n        \n        train_fake_videos = fake_videos[:train_count]\n        val_fake_videos = fake_videos[train_count:]\n        \n        # Process each subset\n        for video in tqdm(train_real_videos, desc=\"Processing train real videos\"):\n            video_dir = os.path.join(ff_real_dir, video)\n            if os.path.isdir(video_dir):\n                # Process every 10th frame to reduce dataset size\n                frames = [f for f in os.listdir(video_dir) if f.endswith('.png')]\n                frames = sorted(frames)[::10]\n                for frame in frames:\n                    img_path = os.path.join(video_dir, frame)\n                    faces = extract_faces(img_path, mtcnn, real_output_dir)\n                    if faces:\n                        all_processed_data['train']['real'].extend(faces)\n        \n        for video in tqdm(train_fake_videos, desc=\"Processing train fake videos\"):\n            video_dir = os.path.join(ff_fake_dir, video)\n            if os.path.isdir(video_dir):\n                frames = [f for f in os.listdir(video_dir) if f.endswith('.png')]\n                frames = sorted(frames)[::10]\n                for frame in frames:\n                    img_path = os.path.join(video_dir, frame)\n                    faces = extract_faces(img_path, mtcnn, fake_output_dir)\n                    if faces:\n                        all_processed_data['train']['fake'].extend(faces)\n                        \n        # Process validation data\n        for video in tqdm(val_real_videos, desc=\"Processing val real videos\"):\n            video_dir = os.path.join(ff_real_dir, video)\n            if os.path.isdir(video_dir):\n                frames = [f for f in os.listdir(video_dir) if f.endswith('.png')]\n                frames = sorted(frames)[::10]\n                for frame in frames:\n                    img_path = os.path.join(video_dir, frame)\n                    faces = extract_faces(img_path, mtcnn, real_output_dir)\n                    if faces:\n                        all_processed_data['val']['real'].extend(faces)\n        \n        for video in tqdm(val_fake_videos, desc=\"Processing val fake videos\"):\n            video_dir = os.path.join(ff_fake_dir, video)\n            if os.path.isdir(video_dir):\n                frames = [f for f in os.listdir(video_dir) if f.endswith('.png')]\n                frames = sorted(frames)[::10]\n                for frame in frames:\n                    img_path = os.path.join(video_dir, frame)\n                    faces = extract_faces(img_path, mtcnn, fake_output_dir)\n                    if faces:\n                        all_processed_data['val']['fake'].extend(faces)\n    \n    # Process other datasets if available (Celeb-DF, DFDC, etc.)\n    \n    # Create training and validation sets\n    train_paths = all_processed_data['train']['real'] + all_processed_data['train']['fake']\n    train_labels = [0] * len(all_processed_data['train']['real']) + [1] * len(all_processed_data['train']['fake'])\n    \n    val_paths = all_processed_data['val']['real'] + all_processed_data['val']['fake']\n    val_labels = [0] * len(all_processed_data['val']['real']) + [1] * len(all_processed_data['val']['fake'])\n    \n    # Shuffle the data\n    train_data = list(zip(train_paths, train_labels))\n    val_data = list(zip(val_paths, val_labels))\n    \n    random.shuffle(train_data)\n    random.shuffle(val_data)\n    \n    train_paths, train_labels = zip(*train_data) if train_data else ([], [])\n    val_paths, val_labels = zip(*val_data) if val_data else ([], [])\n    \n    return train_paths, train_labels, val_paths, val_labels\n\n# Run data processing pipeline\nprint(\"Starting data processing pipeline...\")\ntry:\n    train_paths, train_labels, val_paths, val_labels = process_datasets()\n    print(f\"Processed {len(train_paths)} training images and {len(val_paths)} validation images\")\n    \n    # Save paths and labels for future use\n    np.save('/kaggle/working/train_paths.npy', train_paths)\n    np.save('/kaggle/working/train_labels.npy', train_labels)\n    np.save('/kaggle/working/val_paths.npy', val_paths)\n    np.save('/kaggle/working/val_labels.npy', val_labels)\n    \nexcept Exception as e:\n    print(f\"Error in data processing: {str(e)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Enhanced Dataset with Spectral Features","metadata":{}},{"cell_type":"code","source":"class SpectralDeepfakeDataset(Dataset):\n    \"\"\"Dataset class that combines RGB images with spectral features\"\"\"\n    def __init__(self, image_paths, labels, transform=None):\n        self.image_paths = image_paths\n        self.labels = labels\n        self.transform = transform\n        \n        # Create spectral features directory\n        self.spectral_dir = \"/kaggle/working/spectral_features\"\n        os.makedirs(self.spectral_dir, exist_ok=True)\n        \n        # Pre-compute spectral features for all images\n        self.spectral_paths = self._precompute_spectral_features()\n        \n    def _precompute_spectral_features(self):\n        spectral_paths = []\n        for img_path in tqdm(self.image_paths, desc=\"Computing spectral features\"):\n            base_name = os.path.splitext(os.path.basename(img_path))[0]\n            spectral_path = os.path.join(self.spectral_dir, f\"{base_name}_spectral.png\")\n            \n            # Check if already computed\n            if not os.path.exists(spectral_path):\n                spectral = extract_spectral_features(img_path)\n                if spectral is not None:\n                    cv2.imwrite(spectral_path, (spectral * 255).astype(np.uint8))\n            \n            spectral_paths.append(spectral_path)\n        return spectral_paths\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, idx):\n        # Load RGB image\n        img_path = self.image_paths[idx]\n        image = Image.open(img_path).convert('RGB')\n        \n        # Load spectral feature\n        spectral_path = self.spectral_paths[idx]\n        if os.path.exists(spectral_path):\n            spectral = Image.open(spectral_path).convert('L')  # Grayscale\n        else:\n            # Create blank spectral image if not found\n            spectral = Image.new('L', image.size, 0)\n            \n        label = self.labels[idx]\n        \n        # Apply transform to both\n        if self.transform:\n            # Make sure transformations are applied consistently to both images\n            seed = np.random.randint(2147483647)\n            \n            random.seed(seed)\n            torch.manual_seed(seed)\n            image = self.transform(image)\n            \n            random.seed(seed)\n            torch.manual_seed(seed)\n            spectral = self.transform(spectral)\n        \n        # Return RGB image, spectral feature, and label\n        return image, spectral, label\n\ndef create_spectral_data_loaders(train_paths, train_labels, val_paths, val_labels, batch_size=32):\n    \"\"\"Create data loaders with spectral features\"\"\"\n    # Define transformations for training and validation\n    train_transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.RandomHorizontalFlip(),\n        transforms.ColorJitter(brightness=0.1, contrast=0.1, saturation=0.1),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    valid_transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    # Create spectral datasets\n    train_dataset = SpectralDeepfakeDataset(train_paths, train_labels, train_transform)\n    valid_dataset = SpectralDeepfakeDataset(val_paths, val_labels, valid_transform)\n    \n    # Create data loaders\n    train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=2, pin_memory=True)\n    valid_loader = DataLoader(valid_dataset, batch_size=batch_size, shuffle=False, num_workers=2, pin_memory=True)\n    \n    return train_loader, valid_loader","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Adversarial Augmentation","metadata":{}},{"cell_type":"code","source":"class AdversarialAugmentation:\n    \"\"\"Class for generating adversarial deepfake variations\"\"\"\n    \n    def __init__(self, device='cuda'):\n        self.device = device\n        \n    def alpha_blend(self, real_face, fake_face, alpha=0.5):\n        \"\"\"Alpha blending between real and fake faces\"\"\"\n        blended = alpha * real_face + (1 - alpha) * fake_face\n        return blended\n    \n    def poisson_blend(self, real_face, fake_face, mask=None):\n        \"\"\"Poisson blending between real and fake faces\"\"\"\n        # Convert tensors to numpy for OpenCV\n        if isinstance(real_face, torch.Tensor):\n            real_face = real_face.permute(1, 2, 0).cpu().numpy()\n            real_face = ((real_face * 0.5 + 0.5) * 255).astype(np.uint8)\n            \n        if isinstance(fake_face, torch.Tensor):\n            fake_face = fake_face.permute(1, 2, 0).cpu().numpy()\n            fake_face = ((fake_face * 0.5 + 0.5) * 255).astype(np.uint8)\n            \n        # Create mask if not provided\n        if mask is None:\n            h, w = real_face.shape[:2]\n            center = (w // 2, h // 2)\n            radius = min(h, w) // 3\n            mask = np.zeros((h, w), dtype=np.uint8)\n            cv2.circle(mask, center, radius, 255, -1)\n        \n        # Perform Poisson blending\n        blended = cv2.seamlessClone(fake_face, real_face, mask, (real_face.shape[1]//2, real_face.shape[0]//2), cv2.MIXED_CLONE)\n        \n        # Convert back to tensor format\n        blended = blended.astype(np.float32) / 255.0\n        blended = blended * 2 - 1  # Scale to [-1, 1]\n        blended = torch.from_numpy(blended).permute(2, 0, 1)\n        \n        return blended\n        \n    def mixup(self, real_face, fake_face, alpha=0.2):\n        \"\"\"Mixup augmentation between real and fake\"\"\"\n        lam = np.random.beta(alpha, alpha)\n        mixed = lam * real_face + (1 - lam) * fake_face\n        return mixed, lam\n    \n    def generate_batch_variations(self, real_batch, fake_batch, num_variations=2):\n        \"\"\"Generate a batch of adversarial deepfake variations\"\"\"\n        batch_size = real_batch.size(0)\n        variations = []\n        variation_types = []\n        \n        for i in range(batch_size):\n            real_face = real_batch[i]\n            fake_face = fake_batch[i % fake_batch.size(0)]  # Cycle through fake batch if sizes don't match\n            \n            # Randomly select augmentation type\n            aug_type = random.choice(['alpha', 'mixup'])  # poisson is computationally expensive\n            \n            if aug_type == 'alpha':\n                alpha = random.uniform(0.2, 0.8)\n                variation = self.alpha_blend(real_face, fake_face, alpha)\n                variation_types.append([alpha, 0.0])  # [alpha, lam]\n            \n            elif aug_type == 'mixup':\n                variation, lam = self.mixup(real_face, fake_face)\n                variation_types.append([0.0, lam])  # [alpha, lam]\n                \n            variations.append(variation)\n        \n        return torch.stack(variations), torch.tensor(variation_types, device=self.device)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Architecture with Self-Supervised Learning","metadata":{}},{"cell_type":"code","source":"class DeepfakeDetector(nn.Module):\n    \"\"\"Deepfake detector with self-supervised learning capabilities\"\"\"\n    \n    def __init__(self, backbone='efficientnet-b0', pretrained=True):\n        super(DeepfakeDetector, self).__init__()\n        \n        # Backbone feature extractor\n        if backbone.startswith('efficientnet'):\n            self.backbone = EfficientNet.from_pretrained(backbone) if pretrained else EfficientNet.from_name(backbone)\n            feature_dim = self.backbone._fc.in_features\n            self.backbone._fc = nn.Identity()\n        elif backbone == 'resnet50':\n            self.backbone = models.resnet50(pretrained=pretrained)\n            feature_dim = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n        elif backbone == 'vit':\n            self.backbone = models.vit_b_16(pretrained=pretrained)\n            feature_dim = self.backbone.heads.head.in_features\n            self.backbone.heads.head = nn.Identity()\n        else:\n            raise ValueError(f\"Unsupported backbone: {backbone}\")\n            \n        # Spectral feature extractor\n        self.spectral_processor = nn.Sequential(\n            nn.Conv2d(1, 16, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n            nn.Conv2d(16, 32, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n            nn.Conv2d(32, 64, kernel_size=3, padding=1),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool2d(1),\n            nn.Flatten()\n        )\n        spectral_feature_dim = 64\n        \n        # Main classifier - binary classification (real/fake)\n        self.classifier = nn.Sequential(\n            nn.Linear(feature_dim + spectral_feature_dim, 256),\n            nn.ReLU(),\n            nn.Dropout(0.5),\n            nn.Linear(256, 1),\n            nn.Sigmoid()\n        )\n        \n        # Self-supervised forgery configuration predictor\n        self.forgery_config_predictor = nn.Sequential(\n            nn.Linear(feature_dim + spectral_feature_dim, 128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(128, 2)  # For alpha and lambda values\n        )\n        \n        # Additional feature for supervision signal: deepfake type classifier\n        self.deepfake_type_classifier = nn.Sequential(\n            nn.Linear(feature_dim + spectral_feature_dim, 128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(128, 5)  # 5 deepfake types (Deepfakes, Face2Face, FaceSwap, etc.)\n        )\n        \n    def forward(self, x, spectral=None, mode='classification'):\n        \"\"\"Forward pass\n        Args:\n            x: RGB image input\n            spectral: Spectral feature input (optional)\n            mode: Operation mode ('classification', 'self_supervised', 'representation')\n        \"\"\"\n        # Extract features from RGB image\n        features = self.backbone(x)\n        \n        # Process spectral features if provided\n        if spectral is not None:\n            spectral_features = self.spectral_processor(spectral)\n            # Combine RGB and spectral features\n            combined_features = torch.cat((features, spectral_features), dim=1)\n        else:\n            combined_features = features\n        \n        # Return different outputs based on mode\n        if mode == 'classification':\n            return self.classifier(combined_features)\n        elif mode == 'self_supervised':\n            return {\n                'classification': self.classifier(combined_features),\n                'forgery_config': self.forgery_config_predictor(combined_features),\n                'deepfake_type': self.deepfake_type_classifier(combined_features)\n            }\n        elif mode == 'representation':\n            return combined_features\n        else:\n            raise ValueError(f\"Unsupported mode: {mode}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Continual Learning Implementation","metadata":{}},{"cell_type":"code","source":"class ContinualLearningManager:\n    \"\"\"Manager for continual learning of the deepfake detector\"\"\"\n    \n    def __init__(self, model, device, memory_size=200, strategy='icarl'):\n        \"\"\"\n        Initialize the continual learning manager\n        Args:\n            model: DeepfakeDetector model\n            device: Device to run the model on\n            memory_size: Size of the memory buffer\n            strategy: Continual learning strategy ('icarl', 'gdumb', 'lwf')\n        \"\"\"\n        self.model = model\n        self.device = device\n        self.memory_size = memory_size\n        self.strategy = strategy\n        \n        # Memory buffer for exemplars\n        self.memory_buffer = {'data': [], 'labels': [], 'features': []}\n        \n        # Previous model for knowledge distillation (LwF)\n        self.previous_model = None\n        \n    def update_memory_buffer(self, dataloader, task_id=0):\n        \"\"\"Update memory buffer with new examples using herding selection\"\"\"\n        if self.strategy not in ['icarl', 'gdumb']:\n            return\n            \n        # Extract features from the current dataset\n        features = []\n        data_items = []\n        labels = []\n        \n        self.model.eval()\n        with torch.no_grad():\n            for images, spectrals, targets in tqdm(dataloader, desc=\"Extracting features for memory\"):\n                images = images.to(self.device)\n                spectrals = spectrals.to(self.device)\n                \n                batch_features = self.model(images, spectrals, mode='representation')\n                \n                features.append(batch_features.cpu())\n                data_items.extend(list(zip(images.cpu(), spectrals.cpu())))\n                labels.extend(targets.cpu().numpy())\n        \n        features = torch.cat(features, dim=0)\n        \n        # Separate real and fake examples\n        real_indices = [i for i, label in enumerate(labels) if label == 0]\n        fake_indices = [i for i, label in enumerate(labels) if label == 1]\n        \n        # Calculate class-wise memory size\n        memory_per_class = self.memory_size // 2\n        \n        # Perform herding selection for real examples\n        if len(real_indices) > 0:\n            real_features = features[real_indices]\n            real_mean = torch.mean(real_features, dim=0)\n            \n            selected_real_indices = []\n            current_mean = torch.zeros_like(real_mean)\n            \n            for _ in range(min(memory_per_class, len(real_indices))):\n                # Calculate distance from current_mean to real_mean for each example\n                distances = torch.norm(\n                    (current_mean / (_ + 1) if _ > 0 else 0) + \n                    real_features / (_ + 1) - \n                    real_mean.unsqueeze(0), \n                    dim=1\n                )\n                \n                # Select the example that minimizes this distance\n                idx = torch.argmin(distances)\n                selected_real_indices.append(real_indices[idx])\n                \n                # Update current mean\n                current_mean += real_features[idx]\n                \n                # Remove the selected example\n                real_features = torch.cat([real_features[:idx], real_features[idx+1:]], dim=0)\n                real_indices.pop(idx)\n        else:\n            selected_real_indices = []\n        \n        # Perform herding selection for fake examples (same approach)\n        if len(fake_indices) > 0:\n            fake_features = features[fake_indices]\n            fake_mean = torch.mean(fake_features, dim=0)\n            \n            selected_fake_indices = []\n            current_mean = torch.zeros_like(fake_mean)\n            \n            for _ in range(min(memory_per_class, len(fake_indices))):\n                distances = torch.norm(\n                    (current_mean / (_ + 1) if _ > 0 else 0) + \n                    fake_features / (_ + 1) - \n                    fake_mean.unsqueeze(0), \n                    dim=1\n                )\n                \n                idx = torch.argmin(distances)\n                selected_fake_indices.append(fake_indices[idx])\n                \n                current_mean += fake_features[idx]\n                \n                fake_features = torch.cat([fake_features[:idx], fake_features[idx+1:]], dim=0)\n                fake_indices.pop(idx)\n        else:\n            selected_fake_indices = []\n        \n        # Combine selected indices\n        selected_indices = selected_real_indices + selected_fake_indices\n        \n        # Update memory buffer\n        for i in selected_indices:\n            self.memory_buffer['data'].append(data_items[i])\n            self.memory_buffer['labels'].append(labels[i])\n            self.memory_buffer['features'].append(features[i])\n        \n        # Trim memory buffer if it exceeds memory_size\n        if len(self.memory_buffer['data']) > self.memory_size:\n            self.memory_buffer['data'] = self.memory_buffer['data'][-self.memory_size:]\n            self.memory_buffer['labels'] = self.memory_buffer['labels'][-self.memory_size:]\n            self.memory_buffer['features'] = self.memory_buffer['features'][-self.memory_size:]\n    \n    def create_memory_loader(self, batch_size=32):\n        \"\"\"Create a dataloader from the memory buffer\"\"\"\n        if len(self.memory_buffer['data']) == 0:\n            return None\n            \n        memory_images = [item[0] for item in self.memory_buffer['data']]\n        memory_spectrals = [item[1] for item in self.memory_buffer['data']]\n        memory_labels = self.memory_buffer['labels']\n        \n        # Stack tensors\n        memory_images = torch.stack(memory_images)\n        memory_spectrals = torch.stack(memory_spectrals)\n        memory_labels = torch.tensor(memory_labels)\n        \n        # Create TensorDataset and DataLoader\n        memory_dataset = torch.utils.data.TensorDataset(memory_images, memory_spectrals, memory_labels)\n        memory_loader = torch.utils.data.DataLoader(\n            memory_dataset, batch_size=batch_size, shuffle=True\n        )\n        \n        return memory_loader\n    \n    def save_previous_model(self):\n        \"\"\"Save a copy of the current model for knowledge distillation\"\"\"\n        if self.strategy == 'lwf':\n            self.previous_model = copy.deepcopy(self.model)\n            self.previous_model.eval()  # Set to evaluation mode\n    \n    def knowledge_distillation_loss(self, outputs, images, spectrals, temperature=2.0):\n        \"\"\"Calculate knowledge distillation loss between current and previous model\"\"\"\n        if self.previous_model is None or self.strategy != 'lwf':\n            return 0.0\n            \n        # Get outputs from previous model\n        with torch.no_grad():\n            prev_outputs = self.previous_model(images, spectrals)\n            \n        # Apply temperature scaling\n        scaled_outputs = outputs / temperature\n        scaled_prev_outputs = prev_outputs / temperature\n            \n        # Calculate distillation loss (KL divergence)\n        kd_loss = nn.KLDivLoss(reduction='batchmean')(\n            F.log_softmax(scaled_outputs, dim=1),\n            F.softmax(scaled_prev_outputs, dim=1)\n        ) * (temperature ** 2)\n        \n        return kd_loss","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Functions with Self-Supervised and Continual Learning","metadata":{}},{"cell_type":"code","source":"def train_epoch(model, optimizer, train_loader, device, adv_augmentor, cl_manager=None, epoch=0, alpha=0.5, beta=0.1):\n    \"\"\"Train the model for one epoch with self-supervised learning\"\"\"\n    model.train()\n    running_loss = 0.0\n    correct = 0\n    total = 0\n    \n    # Get memory loader for replay\n    memory_loader_iter = None\n    if cl_manager is not None and cl_manager.strategy in ['icarl', 'gdumb']:\n        memory_loader = cl_manager.create_memory_loader()\n        if memory_loader:\n            memory_loader_iter = iter(memory_loader)\n    \n    for i, (images, spectrals, targets) in enumerate(tqdm(train_loader, desc=f\"Training Epoch {epoch}\")):\n        images, spectrals, targets = images.to(device), spectrals.to(device), targets.to(device).float().unsqueeze(1)\n        \n        # Separate real and fake samples\n        real_indices = targets == 0\n        fake_indices = targets == 1\n        \n        real_images = images[real_indices] if torch.any(real_indices) else None\n        fake_images = images[fake_indices] if torch.any(fake_indices) else None\n        \n        real_spectrals = spectrals[real_indices] if torch.any(real_indices) else None\n        fake_spectrals = spectrals[fake_indices] if torch.any(fake_indices) else None\n        \n        # Generate adversarial deepfake variations\n        if real_images is not None and fake_images is not None:\n            adv_images, adv_configs = adv_augmentor.generate_batch_variations(real_images, fake_images)\n            adv_spectrals = torch.zeros((adv_images.size(0), 1, 224, 224), device=device)\n            \n            # Extract spectral features for adversarial examples\n            for j in range(adv_images.size(0)):\n                # Convert tensor to numpy image\n                adv_img = adv_images[j].permute(1, 2, 0).cpu().numpy()\n                adv_img = ((adv_img * 0.5 + 0.5) * 255).astype(np.uint8)\n                \n                # Extract spectral features\n                spectral = extract_spectral_features(adv_img)\n                if spectral is not None:\n                    spectral_tensor = torch.from_numpy(spectral).unsqueeze(0)\n                    adv_spectrals[j] = spectral_tensor.to(device)\n        else:\n            adv_images = None\n            adv_configs = None\n            adv_spectrals = None\n        \n        # Sample from memory buffer if available\n        if memory_loader_iter is not None:\n            try:\n                mem_images, mem_spectrals, mem_targets = next(memory_loader_iter)\n                mem_images, mem_spectrals, mem_targets = mem_images.to(device), mem_spectrals.to(device), mem_targets.to(device).float().unsqueeze(1)\n                \n                # Combine current batch with memory samples\n                images = torch.cat([images, mem_images], dim=0)\n                spectrals = torch.cat([spectrals, mem_spectrals], dim=0)\n                targets = torch.cat([targets, mem_targets], dim=0)\n            except StopIteration:\n                pass\n        \n        # Zero gradients\n        optimizer.zero_grad()\n        \n        # Forward pass for binary classification\n        outputs = model(images, spectrals)\n        \n        # Binary classification loss\n        bce_loss = F.binary_cross_entropy(outputs, targets)\n        loss = bce_loss\n        \n        # Self-supervised learning with adversarial examples\n        if adv_images is not None:\n            # Forward pass for adversarial examples\n            adv_outputs = model(adv_images, adv_spectrals, mode='self_supervised')\n            \n            # Classification loss for adversarial examples (all are fake)\n            adv_targets = torch.ones(adv_images.size(0), 1, device=device)\n            adv_cls_loss = F.binary_cross_entropy(adv_outputs['classification'], adv_targets)\n            \n            # Forgery configuration prediction loss (MSE)\n            adv_config_loss = F.mse_loss(adv_outputs['forgery_config'], adv_configs)\n            \n            # Combine losses\n            self_supervised_loss = adv_cls_loss + alpha * adv_config_loss\n            loss += beta * self_supervised_loss\n        \n        # Knowledge distillation loss for continual learning\n        if cl_manager is not None and cl_manager.strategy == 'lwf' and cl_manager.previous_model is not None:\n            kd_loss = cl_manager.knowledge_distillation_loss(outputs, images, spectrals)\n            loss += 0.5 * kd_loss  # Lambda=0.5 for LwF\n        \n        # Backpropagation\n        loss.backward()\n        optimizer.step()\n        \n        # Statistics\n        running_loss += loss.item()\n        predicted = (outputs > 0.5).float()\n        total += targets.size(0)\n        correct += (predicted == targets).sum().item()\n    \n    # Calculate epoch statistics\n    epoch_loss = running_loss / len(train_loader)\n    epoch_acc = 100.0 * correct / total\n    \n    return epoch_loss, epoch_acc\n\ndef validate(model, val_loader, device):\n    \"\"\"Validate the model\"\"\"\n    model.eval()\n    running_loss = 0.0\n    all_preds = []\n    all_targets = []\n    \n    with torch.no_grad():\n        for images, spectrals, targets in tqdm(val_loader, desc=\"Validation\"):\n            images, spectrals, targets = images.to(device), spectrals.to(device), targets.to(device).float().unsqueeze(1)\n            \n            # Forward pass\n            outputs = model(images, spectrals)\n            \n            # Binary classification loss\n            loss = F.binary_cross_entropy(outputs, targets)\n            \n            # Statistics\n            running_loss += loss.item()\n            all_preds.append(outputs.cpu().numpy())\n            all_targets.append(targets.cpu().numpy())\n    \n    # Concatenate all predictions and targets\n    all_preds = np.concatenate(all_preds).ravel()\n    all_targets = np.concatenate(all_targets).ravel()\n    \n    # Calculate metrics\n    val_loss = running_loss / len(val_loader)\n    auc = roc_auc_score(all_targets, all_preds)\n    \n    # Use 0.5 threshold for accuracy\n    binary_preds = (all_preds > 0.5).astype(int)\n    val_acc = accuracy_score(all_targets, binary_preds)\n    \n    # Calculate mean average precision\n    mAP = average_precision_score(all_targets, all_preds)\n    \n    # Calculate confusion matrix\n    tn, fp, fn, tp = confusion_matrix(all_targets, binary_preds).ravel()\n    val_precision = tp / (tp + fp) if (tp + fp) > 0 else 0\n    val_recall = tp / (tp + fn) if (tp + fn) > 0 else 0\n    val_f1 = 2 * val_precision * val_recall / (val_precision + val_recall) if (val_precision + val_recall) > 0 else 0\n    \n    metrics = {\n        'val_loss': val_loss,\n        'val_acc': val_acc * 100.0,\n        'auc': auc,\n        'mAP': mAP,\n        'precision': val_precision,\n        'recall': val_recall,\n        'f1': val_f1\n    }\n    \n    return metrics","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Full Training Pipeline with Continual Learning","metadata":{}},{"cell_type":"code","source":"def train_deepfake_detector(train_paths, train_labels, val_paths, val_labels, \n                           continual_learning=True, backbone='efficientnet-b0'):\n    \"\"\"Full training pipeline for deepfake detector\"\"\"\n    \n    # Create data loaders\n    print(\"Creating data loaders...\")\n    batch_size = 32\n    train_loader, val_loader = create_spectral_data_loaders(\n        train_paths, train_labels, val_paths, val_labels, batch_size)\n    \n    # Initialize model\n    print(f\"Initializing model with {backbone} backbone...\")\n    model = DeepfakeDetector(backbone=backbone, pretrained=True).to(device)\n    \n    # Initialize adversarial augmentation\n    adv_augmentor = AdversarialAugmentation(device=device)\n    \n    # Initialize continual learning manager if enabled\n    cl_manager = None\n    if continual_learning:\n        cl_manager = ContinualLearningManager(model, device, memory_size=200, strategy='icarl')\n    \n    # Training parameters\n    num_epochs = 20\n    lr = 1e-4\n    optimizer = optim.Adam(model.parameters(), lr=lr)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, 'min', patience=3, factor=0.5)\n    \n    # Training history\n    history = {\n        'train_loss': [],\n        'train_acc': [],\n        'val_loss': [],\n        'val_acc': [],\n        'auc': [],\n        'mAP': []\n    }\n    \n    best_auc = 0.0\n    \n    # Training loop\n    for epoch in range(num_epochs):\n        print(f\"\\nEpoch {epoch+1}/{num_epochs}\")\n        \n        # Save previous model for knowledge distillation\n        if cl_manager is not None and cl_manager.strategy == 'lwf':\n            cl_manager.save_previous_model()\n        \n        # Train one epoch\n        train_loss, train_acc = train_epoch(\n            model, optimizer, train_loader, device, \n            adv_augmentor, cl_manager, epoch)\n        \n        # Validate\n        metrics = validate(model, val_loader, device)\n        \n        # Update learning rate\n        scheduler.step(metrics['val_loss'])\n        \n        # Save history\n        history['train_loss'].append(train_loss)\n        history['train_acc'].append(train_acc)\n        history['val_loss'].append(metrics['val_loss'])\n        history['val_acc'].append(metrics['val_acc'])\n        history['auc'].append(metrics['auc'])\n        history['mAP'].append(metrics['mAP'])\n        \n        # Print metrics\n        print(f\"Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.2f}%\")\n        print(f\"Val Loss: {metrics['val_loss']:.4f}, Val Acc: {metrics['val_acc']:.2f}%, AUC: {metrics['auc']:.4f}, mAP: {metrics['mAP']:.4f}\")\n        print(f\"Precision: {metrics['precision']:.4f}, Recall: {metrics['recall']:.4f}, F1: {metrics['f1']:.4f}\")\n        \n        # Save best model\n        if metrics['auc'] > best_auc:\n            best_auc = metrics['auc']\n            torch.save(model.state_dict(), '/kaggle/working/models/best_model.pth')\n            print(f\"Saved best model with AUC: {best_auc:.4f}\")\n        \n        # Update memory buffer for continual learning\n        if cl_manager is not None:\n            cl_manager.update_memory_buffer(train_loader, epoch)\n    \n    # Save final model\n    torch.save(model.state_dict(), '/kaggle/working/models/final_model.pth')\n    \n    # Save history\n    with open('/kaggle/working/models/history.json', 'w') as f:\n        json.dump(history, f)\n    \n    return model, history\n\n# Trigger the training process\ntry:\n    # Load the preprocessed data (if cell 6 ran successfully)\n    train_paths = np.load('/kaggle/working/train_paths.npy')\n    train_labels = np.load('/kaggle/working/train_labels.npy')\n    val_paths = np.load('/kaggle/working/val_paths.npy')\n    val_labels = np.load('/kaggle/working/val_labels.npy')\n    \n    print(f\"Loaded {len(train_paths)} training examples and {len(val_paths)} validation examples\")\n    \n    # Train the model\n    model, history = train_deepfake_detector(\n        train_paths, train_labels, val_paths, val_labels, \n        continual_learning=True, backbone='efficientnet-b0')\n    \nexcept FileNotFoundError:\n    print(\"Pre-processed data not found. Please run data processing cell first.\")\nexcept Exception as e:\n    print(f\"Error during training: {str(e)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Incremental Learning with New Deepfake Types","metadata":{}},{"cell_type":"code","source":"def incremental_learning_pipeline(model, new_dataset_paths, new_dataset_labels, cl_manager):\n    \"\"\"Incremental learning pipeline for new deepfake types\"\"\"\n    \n    # Create data loaders for new dataset\n    val_split = 0.2\n    split_idx = int(len(new_dataset_paths) * (1 - val_split))\n    \n    train_paths = new_dataset_paths[:split_idx]\n    train_labels = new_dataset_labels[:split_idx]\n    val_paths = new_dataset_paths[split_idx:]\n    val_labels = new_dataset_labels[split_idx:]\n    \n    train_loader, val_loader = create_spectral_data_loaders(\n        train_paths, train_labels, val_paths, val_labels, batch_size=32)\n    \n    # Initialize adversarial augmentation\n    adv_augmentor = AdversarialAugmentation(device=device)\n    \n    # Save previous model for knowledge distillation if using LwF\n    if cl_manager.strategy == 'lwf':\n        cl_manager.save_previous_model()\n    \n    # Training parameters\n    num_epochs = 10  # Fewer epochs for incremental learning\n    lr = 5e-5  # Lower learning rate\n    optimizer = optim.Adam(model.parameters(), lr=lr)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, 'min', patience=2, factor=0.5)\n    \n    # Training history\n    history = {\n        'train_loss': [],\n        'train_acc': [],\n        'val_loss': [],\n        'val_acc': [],\n        'auc': [],\n        'mAP': []\n    }\n    \n    # Training loop\n    for epoch in range(num_epochs):\n        print(f\"\\nIncremental Learning Epoch {epoch+1}/{num_epochs}\")\n        \n        # Train one epoch\n        train_loss, train_acc = train_epoch(\n            model, optimizer, train_loader, device, \n            adv_augmentor, cl_manager, epoch)\n        \n        # Validate\n        metrics = validate(model, val_loader, device)\n        \n        # Update learning rate\n        scheduler.step(metrics['val_loss'])\n        \n        # Save history\n        history['train_loss'].append(train_loss)\n        history['train_acc'].append(train_acc)\n        history['val_loss'].append(metrics['val_loss'])\n        history['val_acc'].append(metrics['val_acc'])\n        history['auc'].append(metrics['auc'])\n        history['mAP'].append(metrics['mAP'])\n        \n        # Print metrics\n        print(f\"Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.2f}%\")\n        print(f\"Val Loss: {metrics['val_loss']:.4f}, Val Acc: {metrics['val_acc']:.2f}%, AUC: {metrics['auc']:.4f}\")\n        \n        # Update memory buffer\n        cl_manager.update_memory_buffer(train_loader, epoch)\n    \n    # Save incremental model\n    timestamp = time.strftime(\"%Y%m%d-%H%M%S\")\n    torch.save(model.state_dict(), f'/kaggle/working/models/incremental_model_{timestamp}.pth')\n    \n    # Save history\n    with open(f'/kaggle/working/models/incremental_history_{timestamp}.json', 'w') as f:\n        json.dump(history, f)\n    \n    return model, history\n\ndef evaluate_catastrophic_forgetting(model, original_val_paths, original_val_labels, \n                                    new_val_paths, new_val_labels):\n    \"\"\"Evaluate catastrophic forgetting by measuring performance on original dataset\"\"\"\n    \n    # Create data loaders\n    original_loader, _ = create_spectral_data_loaders(\n        original_val_paths, original_val_labels, \n        original_val_paths, original_val_labels,  # Same data for both (we'll ignore the second one)\n        batch_size=32)\n    \n    new_loader, _ = create_spectral_data_loaders(\n        new_val_paths, new_val_labels,\n        new_val_paths, new_val_labels,  # Same data for both (we'll ignore the second one)\n        batch_size=32)\n    \n    # Evaluate on original dataset\n    print(\"Evaluating on original dataset:\")\n    original_metrics = validate(model, original_loader, device)\n    \n    # Evaluate on new dataset\n    print(\"Evaluating on new dataset:\")\n    new_metrics = validate(model, new_loader, device)\n    \n    # Calculate forgetting\n    forgetting_rate = {\n        'acc': original_metrics['val_acc'],\n        'auc': original_metrics['auc'],\n        'mAP': original_metrics['mAP'],\n        'new_acc': new_metrics['val_acc'],\n        'new_auc': new_metrics['auc'],\n        'new_mAP': new_metrics['mAP']\n    }\n    \n    print(\"Forgetting Evaluation:\")\n    print(f\"Original Dataset - Acc: {original_metrics['val_acc']:.2f}%, AUC: {original_metrics['auc']:.4f}, mAP: {original_metrics['mAP']:.4f}\")\n    print(f\"New Dataset - Acc: {new_metrics['val_acc']:.2f}%, AUC: {new_metrics['auc']:.4f}, mAP: {new_metrics['mAP']:.4f}\")\n    \n    return forgetting_rate\n\n# Example of using incremental learning with a new dataset\ndef run_incremental_learning_example():\n    \"\"\"Example of running incremental learning with a new deepfake type\"\"\"\n    try:\n        # Load the original model\n        model = DeepfakeDetector(backbone='efficientnet-b0').to(device)\n        model.load_state_dict(torch.load('/kaggle/working/models/best_model.pth'))\n        \n        # Load original validation data for forgetting evaluation\n        val_paths = np.load('/kaggle/working/val_paths.npy')\n        val_labels = np.load('/kaggle/working/val_labels.npy')\n        \n        # Initialize CL manager\n        cl_manager = ContinualLearningManager(model, device, memory_size=200, strategy='icarl')\n        \n        # Load memory buffer with samples from original dataset\n        original_train_paths = np.load('/kaggle/working/train_paths.npy')\n        original_train_labels = np.load('/kaggle/working/train_labels.npy')\n        \n        # Create temporary loader for memory buffer initialization\n        temp_loader, _ = create_spectral_data_loaders(\n            original_train_paths, original_train_labels,\n            val_paths, val_labels,\n            batch_size=32)\n        \n        # Update memory buffer with original dataset samples\n        cl_manager.update_memory_buffer(temp_loader, 0)\n        \n        print(f\"Memory buffer initialized with {len(cl_manager.memory_buffer['data'])} examples\")\n        \n        # Here you would load a new dataset of a different deepfake type\n        # For this example, we'll simulate with a subset of the original data\n        # In a real scenario, you would load a completely new dataset\n        \n        print(\"Simulating a new deepfake type with a subset of original data...\")\n        subset_size = min(1000, len(original_train_paths))\n        new_dataset_paths = original_train_paths[:subset_size]\n        new_dataset_labels = original_train_labels[:subset_size]\n        \n        # Run incremental learning\n        print(\"Starting incremental learning...\")\n        model, history = incremental_learning_pipeline(model, new_dataset_paths, new_dataset_labels, cl_manager)\n        \n        # Evaluate catastrophic forgetting\n        forgetting_rate = evaluate_catastrophic_forgetting(\n            model, val_paths, val_labels, new_dataset_paths[-100:], new_dataset_labels[-100:])\n        \n        # Print summary\n        print(\"\\nIncremental Learning Summary:\")\n        print(f\"Final validation accuracy: {history['val_acc'][-1]:.2f}%\")\n        print(f\"Final validation AUC: {history['auc'][-1]:.4f}\")\n        print(f\"Forgetting evaluation completed.\")\n        \n    except FileNotFoundError:\n        print(\"Required files not found. Please run the main training first.\")\n    except Exception as e:\n        print(f\"Error during incremental learning: {str(e)}\")\n\n# Uncomment to run incremental learning example\n# run_incremental_learning_example()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation and Visualization Functions","metadata":{}},{"cell_type":"code","source":"def evaluate_model(model_path, test_paths, test_labels):\n    \"\"\"Evaluate a trained model on a test set\"\"\"\n    # Load model\n    model = DeepfakeDetector(backbone='efficientnet-b0')\n    model.load_state_dict(torch.load(model_path))\n    model = model.to(device)\n    model.eval()\n    \n    # Create data loader\n    test_transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    test_dataset = SpectralDeepfakeDataset(test_paths, test_labels, test_transform)\n    test_loader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=2, pin_memory=True)\n    \n    # Evaluate\n    metrics = validate(model, test_loader, device)\n    \n    # Print metrics\n    print(f\"Test Results:\")\n    print(f\"Accuracy: {metrics['val_acc']:.2f}%\")\n    print(f\"AUC: {metrics['auc']:.4f}\")\n    print(f\"mAP: {metrics['mAP']:.4f}\")\n    print(f\"Precision: {metrics['precision']:.4f}\")\n    print(f\"Recall: {metrics['recall']:.4f}\")\n    print(f\"F1 Score: {metrics['f1']:.4f}\")\n    \n    # Get predictions for ROC curve\n    all_preds = []\n    all_targets = []\n    \n    with torch.no_grad():\n        for images, spectrals, targets in tqdm(test_loader, desc=\"Getting predictions\"):\n            images, spectrals, targets = images.to(device), spectrals.to(device), targets.to(device)\n            \n            # Forward pass\n            outputs = model(images, spectrals)\n            \n            all_preds.append(outputs.cpu().numpy())\n            all_targets.append(targets.cpu().numpy())\n    \n    # Concatenate all predictions and targets\n    all_preds = np.concatenate(all_preds).ravel()\n    all_targets = np.concatenate(all_targets).ravel()\n    \n    # Plot ROC curve\n    plot_roc_curve(all_targets, all_preds)\n    \n    return metrics, all_preds, all_targets\n\ndef plot_training_history(history_path):\n    \"\"\"Plot training history from saved JSON file\"\"\"\n    with open(history_path, 'r') as f:\n        history = json.load(f)\n    \n    epochs = range(1, len(history['train_loss']) + 1)\n    \n    plt.figure(figsize=(15, 5))\n    \n    # Plot training & validation loss\n    plt.subplot(1, 3, 1)\n    plt.plot(epochs, history['train_loss'], 'b-', label='Training Loss')\n    plt.plot(epochs, history['val_loss'], 'r-', label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    # Plot training & validation accuracy\n    plt.subplot(1, 3, 2)\n    plt.plot(epochs, history['train_acc'], 'b-', label='Training Accuracy')\n    plt.plot(epochs, history['val_acc'], 'r-', label='Validation Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy (%)')\n    plt.legend()\n    \n    # Plot AUC and mAP\n    plt.subplot(1, 3, 3)\n    plt.plot(epochs, history['auc'], 'g-', label='AUC')\n    plt.plot(epochs, history['mAP'], 'm-', label='mAP')\n    plt.title('AUC and mAP')\n    plt.xlabel('Epochs')\n    plt.ylabel('Score')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig('/kaggle/working/training_history.png')\n    plt.show()\n\ndef plot_roc_curve(y_true, y_pred):\n    \"\"\"Plot ROC curve\"\"\"\n    fpr, tpr, _ = roc_curve(y_true, y_pred)\n    roc_auc = auc(fpr, tpr)\n    \n    plt.figure(figsize=(8, 8))\n    plt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (AUC = {roc_auc:.4f})')\n    plt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic (ROC) Curve')\n    plt.legend(loc=\"lower right\")\n    plt.savefig('/kaggle/working/roc_curve.png')\n    plt.show()\n    \n    return fpr, tpr, roc_auc\n\ndef visualize_model_predictions(model_path, test_paths, test_labels, num_samples=10):\n    \"\"\"Visualize model predictions on random test samples\"\"\"\n    # Load model\n    model = DeepfakeDetector(backbone='efficientnet-b0')\n    model.load_state_dict(torch.load(model_path))\n    model = model.to(device)\n    model.eval()\n    \n    # Select random samples\n    indices = np.random.choice(range(len(test_paths)), num_samples, replace=False)\n    sample_paths = [test_paths[i] for i in indices]\n    sample_labels = [test_labels[i] for i in indices]\n    \n    # Transform for visualization\n    test_transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    # Create figure\n    plt.figure(figsize=(20, 4 * (num_samples + 1) // 5))\n    \n    for i, (img_path, label) in enumerate(zip(sample_paths, sample_labels)):\n        # Load image\n        img = Image.open(img_path).convert('RGB')\n        \n        # Create spectral feature\n        spectral_feature = extract_spectral_features(img_path)\n        if spectral_feature is None:\n            spectral_tensor = torch.zeros((1, 224, 224))\n        else:\n            spectral_tensor = torch.from_numpy(\n                cv2.resize(spectral_feature, (224, 224))\n            ).unsqueeze(0)\n        \n        # Transform for model input\n        img_tensor = test_transform(img).unsqueeze(0).to(device)\n        spectral_tensor = test_transform(\n            Image.fromarray((spectral_feature * 255).astype(np.uint8))\n        ).unsqueeze(0).to(device)\n        \n        # Get prediction\n        with torch.no_grad():\n            pred = model(img_tensor, spectral_tensor).item()\n        \n        # Convert prediction to binary\n        pred_class = 1 if pred > 0.5 else 0\n        \n        # Calculate prediction confidence\n        confidence = pred if pred_class == 1 else 1 - pred\n        \n        # Plot image\n        plt.subplot(((num_samples + 4) // 5), 5, i + 1)\n        plt.imshow(np.asarray(img))\n        title = f\"True: {'Fake' if label == 1 else 'Real'}\\n\"\n        title += f\"Pred: {'Fake' if pred_class == 1 else 'Real'} ({confidence:.2f})\"\n        title += f\"\\n{'✓' if pred_class == label else '✗'}\"\n        plt.title(title, color='green' if pred_class == label else 'red')\n        plt.axis('off')\n    \n    plt.tight_layout()\n    plt.savefig('/kaggle/working/model_predictions.png')\n    plt.show()\n\ndef visualize_activation_maps(model_path, test_paths, test_labels, num_samples=5):\n    \"\"\"Visualize activation maps (attention) on random test samples\"\"\"\n    # Load model\n    model = DeepfakeDetector(backbone='efficientnet-b0')\n    model.load_state_dict(torch.load(model_path))\n    model = model.to(device)\n    model.eval()\n    \n    # Select random samples\n    indices = np.random.choice(range(len(test_paths)), num_samples, replace=False)\n    sample_paths = [test_paths[i] for i in indices]\n    sample_labels = [test_labels[i] for i in indices]\n    \n    # Create hook for getting features\n    activation = {}\n    def get_activation(name):\n        def hook(model, input, output):\n            activation[name] = output.detach()\n        return hook\n    \n    # Register hook to get features from the last convolutional layer\n    model.backbone._blocks[-1].register_forward_hook(get_activation('features'))\n    \n    # Create figure\n    plt.figure(figsize=(15, 3*num_samples))\n    \n    for i, (img_path, label) in enumerate(zip(sample_paths, sample_labels)):\n        # Load and preprocess image\n        img = Image.open(img_path).convert('RGB')\n        img_tensor = transforms.Compose([\n            transforms.Resize((224, 224)),\n            transforms.ToTensor(),\n            transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n        ])(img).unsqueeze(0).to(device)\n        \n        # Create spectral feature\n        spectral_feature = extract_spectral_features(img_path)\n        if spectral_feature is None:\n            spectral_img = np.zeros((224, 224))\n        else:\n            spectral_img = cv2.resize(spectral_feature, (224, 224))\n            \n        spectral_tensor = transforms.Compose([\n            transforms.ToTensor(),\n            transforms.Normalize(mean=[0.485], std=[0.229])\n        ])(Image.fromarray((spectral_img * 255).astype(np.uint8))).unsqueeze(0).to(device)\n        \n        # Forward pass\n        with torch.no_grad():\n            pred = model(img_tensor, spectral_tensor).item()\n        \n        # Get activation map\n        features = activation['features'].squeeze().mean(dim=0).cpu().numpy()\n        \n        # Resize to image dimensions\n        heatmap = cv2.resize(features, (224, 224))\n        \n        # Normalize heatmap\n        heatmap = (heatmap - heatmap.min()) / (heatmap.max() - heatmap.min() + 1e-8)\n        \n        # Convert to RGB heatmap\n        heatmap = cv2.applyColorMap((heatmap * 255).astype(np.uint8), cv2.COLORMAP_JET)\n        heatmap = cv2.cvtColor(heatmap, cv2.COLOR_BGR2RGB)\n        \n        # Original image (unnormalized)\n        img_np = np.array(img.resize((224, 224)))\n        \n        # Overlay heatmap on original image\n        overlaid = cv2.addWeighted(img_np, 0.7, heatmap, 0.3, 0)\n        \n        # Display\n        plt.subplot(num_samples, 3, i*3 + 1)\n        plt.imshow(img_np)\n        plt.title(f\"Original {'Fake' if label == 1 else 'Real'}\")\n        plt.axis('off')\n        \n        plt.subplot(num_samples, 3, i*3 + 2)\n        plt.imshow(heatmap)\n        plt.title(\"Activation Map\")\n        plt.axis('off')\n        \n        plt.subplot(num_samples, 3, i*3 + 3)\n        plt.imshow(overlaid)\n        plt.title(f\"Overlay (Pred: {'Fake' if pred > 0.5 else 'Real'}, {pred:.2f})\")\n        plt.axis('off')\n    \n    plt.tight_layout()\n    plt.savefig('/kaggle/working/activation_maps.png')\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Deployment with ONNX and FastAPI","metadata":{}},{"cell_type":"code","source":"def convert_to_onnx(model_path, output_path='/kaggle/working/deepfake_detector.onnx'):\n    \"\"\"Convert PyTorch model to ONNX format for deployment\"\"\"\n    # Load model\n    model = DeepfakeDetector(backbone='efficientnet-b0')\n    model.load_state_dict(torch.load(model_path))\n    model.eval()\n    \n    # Create dummy input\n    dummy_img = torch.randn(1, 3, 224, 224)\n    dummy_spectral = torch.randn(1, 1, 224, 224)\n    \n    # Export to ONNX\n    torch.onnx.export(\n        model,\n        (dummy_img, dummy_spectral),\n        output_path,\n        export_params=True,\n        opset_version=12,\n        do_constant_folding=True,\n        input_names=['image', 'spectral'],\n        output_names=['output'],\n        dynamic_axes={\n            'image': {0: 'batch_size'},\n            'spectral': {0: 'batch_size'},\n            'output': {0: 'batch_size'}\n        }\n    )\n    \n    print(f\"Model exported to {output_path}\")\n    \n    # Verify the exported model\n    import onnx\n    onnx_model = onnx.load(output_path)\n    onnx.checker.check_model(onnx_model)\n    print(\"ONNX model checked - No errors found\")\n    \n    return output_path\n\n# FastAPI code (not executed in Kaggle notebook, but ready for deployment)\n\"\"\"\n# FastAPI deployment code\nfrom fastapi import FastAPI, File, UploadFile\nimport io\nimport numpy as np\nimport onnxruntime\nfrom PIL import Image\nimport cv2\nimport uvicorn\n\napp = FastAPI(title=\"Deepfake Detector API\", description=\"API for detecting deepfake images\")\n\n# Load the ONNX model\nonnx_model_path = \"deepfake_detector.onnx\"\nsession = onnxruntime.InferenceSession(onnx_model_path)\n\n@app.post(\"/detect/\")\nasync def detect_deepfake(file: UploadFile = File(...)):\n    # Read and process the image\n    contents = await file.read()\n    image = Image.open(io.BytesIO(contents)).convert(\"RGB\")\n    image = image.resize((224, 224))\n    \n    # Preprocess the image\n    img_array = np.array(image)\n    \n    # Extract spectral features\n    gray = cv2.cvtColor(img_array, cv2.COLOR_RGB2GRAY)\n    dft = cv2.dft(np.float32(gray), flags=cv2.DFT_COMPLEX_OUTPUT)\n    dft_shift = np.fft.fftshift(dft)\n    magnitude_spectrum = 20 * np.log(cv2.magnitude(dft_shift[:,:,0], dft_shift[:,:,1]) + 1)\n    \n    # Normalize\n    magnitude_spectrum_norm = (magnitude_spectrum - np.min(magnitude_spectrum)) / \\\n                            (np.max(magnitude_spectrum) - np.min(magnitude_spectrum) + 1e-8)\n    \n    # Prepare inputs\n    img_input = img_array.transpose(2, 0, 1).astype(np.float32) / 255.0\n    img_input = (img_input - np.array([0.485, 0.456, 0.406])[:, None, None]) / np.array([0.229, 0.224, 0.225])[:, None, None]\n    \n    spectral_input = magnitude_spectrum_norm.astype(np.float32).reshape(1, 224, 224)\n    spectral_input = (spectral_input - 0.485) / 0.229\n    \n    # Run inference\n    ort_inputs = {\n        'image': img_input.reshape(1, 3, 224, 224),\n        'spectral': spectral_input.reshape(1, 1, 224, 224)\n    }\n    \n    ort_output = session.run(['output'], ort_inputs)[0]\n    \n    # Process result\n    probability = float(ort_output[0][0])\n    prediction = \"fake\" if probability > 0.5 else \"real\"\n    confidence = probability if prediction == \"fake\" else 1 - probability\n    \n    return {\n        \"prediction\": prediction,\n        \"confidence\": confidence,\n        \"probability_fake\": probability\n    }\n\nif __name__ == \"__main__\":\n    uvicorn.run(app, host=\"0.0.0.0\", port=8000)\n\"\"\"\n\n# Generate example conversion code\ndef export_model_example():\n    \"\"\"Example of exporting the model to ONNX format\"\"\"\n    try:\n        model_path = '/kaggle/working/models/best_model.pth'\n        if os.path.exists(model_path):\n            onnx_path = convert_to_onnx(model_path)\n            print(f\"Model successfully exported to ONNX format at {onnx_path}\")\n            print(\"This ONNX model can be deployed using the FastAPI code provided.\")\n        else:\n            print(\"Model file not found. Train the model first.\")\n    except Exception as e:\n        print(f\"Error exporting model: {str(e)}\")\n\n# Uncomment to export model\n# export_model_example()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Real-time Inference with OpenCV","metadata":{}},{"cell_type":"code","source":"def setup_realtime_detector(model_path):\n    \"\"\"Set up a real-time deepfake detector for video streams or webcam\"\"\"\n    # Load model\n    model = DeepfakeDetector(backbone='efficientnet-b0')\n    model.load_state_dict(torch.load(model_path))\n    model = model.to(device)\n    model.eval()\n    \n    # Initialize face detector\n    mtcnn = MTCNN(keep_all=True, device=device)\n    \n    # Define image transforms\n    transform = transforms.Compose([\n        transforms.Resize((224, 224)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n    ])\n    \n    return model, mtcnn, transform\n\ndef process_video(video_path, model_path, output_path=None):\n    \"\"\"Process a video file and detect deepfakes in each frame\"\"\"\n    # Set up detector\n    model, mtcnn, transform = setup_realtime_detector(model_path)\n    \n    # Open video file\n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(\"Error: Could not open video.\")\n        return\n    \n    # Get video properties\n    frame_width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))\n    frame_height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    \n    # Set up video writer if output path provided\n    writer = None\n    if output_path:\n        fourcc = cv2.VideoWriter_fourcc(*'mp4v')\n        writer = cv2.VideoWriter(output_path, fourcc, fps, (frame_width, frame_height))\n    \n    frame_count = 0\n    processing_times = []\n    \n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n            \n        # Process every 5th frame to improve speed\n        if frame_count % 5 == 0:\n            start_time = time.time()\n            \n            # Convert to RGB for MTCNN\n            frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n            \n            # Detect faces\n            boxes, probs = mtcnn.detect(frame_rgb)\n            \n            if boxes is not None:\n                for i, (box, prob) in enumerate(zip(boxes, probs)):\n                    if prob < 0.9:\n                        continue\n                        \n                    # Extract face with margin\n                    x1, y1, x2, y2 = [int(coord) for coord in box]\n                    \n                    # Skip if face is too small\n                    if (x2 - x1 < 50) or (y2 - y1 < 50):\n                        continue\n                    \n                    # Add margin\n                    h, w = frame.shape[:2]\n                    margin = 20\n                    x1 = max(0, x1 - margin)\n                    y1 = max(0, y1 - margin)\n                    x2 = min(w, x2 + margin)\n                    y2 = min(h, y2 + margin)\n                    \n                    face = frame_rgb[y1:y2, x1:x2]\n                    \n                    if face.size == 0:\n                        continue\n                    \n                    # Extract spectral features\n                    face_gray = cv2.cvtColor(face, cv2.COLOR_RGB2GRAY)\n                    dft = cv2.dft(np.float32(face_gray), flags=cv2.DFT_COMPLEX_OUTPUT)\n                    dft_shift = np.fft.fftshift(dft)\n                    magnitude_spectrum = 20 * np.log(cv2.magnitude(dft_shift[:,:,0], dft_shift[:,:,1]) + 1)\n                    \n                    # Normalize and resize\n                    if magnitude_spectrum.size > 0:\n                        magnitude_spectrum = cv2.resize(magnitude_spectrum, (224, 224))\n                        magnitude_spectrum_norm = (magnitude_spectrum - np.min(magnitude_spectrum)) / \\\n                                                (np.max(magnitude_spectrum) - np.min(magnitude_spectrum) + 1e-8)\n                    else:\n                        magnitude_spectrum_norm = np.zeros((224, 224))\n                    \n                    # Prepare inputs\n                    face_pil = Image.fromarray(face)\n                    face_tensor = transform(face_pil).unsqueeze(0).to(device)\n                    \n                    spectral_pil = Image.fromarray((magnitude_spectrum_norm * 255).astype(np.uint8))\n                    spectral_tensor = transform(spectral_pil).unsqueeze(0).to(device)\n                    \n                    # Get prediction\n                    with torch.no_grad():\n                        pred = model(face_tensor, spectral_tensor).item()\n                    \n                    # Determine color based on prediction\n                    if pred > 0.5:  # Fake\n                        color = (0, 0, 255)  # Red\n                        label = f\"Fake: {pred:.2f}\"\n                    else:  # Real\n                        color = (0, 255, 0)  # Green\n                        label = f\"Real: {1-pred:.2f}\"\n                    \n                    # Draw bounding box and label\n                    cv2.rectangle(frame, (x1, y1), (x2, y2), color, 2)\n                    cv2.putText(frame, label, (x1, y1 - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.9, color, 2)\n            \n            end_time = time.time()\n            processing_times.append(end_time - start_time)\n        \n        # Write the frame if output path provided\n        if writer:\n            writer.write(frame)\n        \n        # Display the frame (if running in an environment that supports it)\n        # cv2.imshow('Deepfake Detection', frame)\n        # if cv2.waitKey(1) & 0xFF == ord('q'):\n        #     break\n        \n        frame_count += 1\n        \n        # Print progress\n        if frame_count % 50 == 0:\n            print(f\"Processed {frame_count} frames\")\n    \n    # Clean up\n    cap.release()\n    if writer:\n        writer.release()\n    # cv2.destroyAllWindows()\n    \n    # Print statistics\n    if processing_times:\n        avg_time = sum(processing_times) / len(processing_times)\n        print(f\"Average processing time per frame: {avg_time:.4f} seconds\")\n        print(f\"Approximate FPS: {1/avg_time:.2f}\")\n    \n    print(f\"Finished processing {frame_count} frames\")\n    if output_path:\n        print(f\"Output saved to {output_path}\")\n\n# Example call (not run in notebook)\n\"\"\"\n# Process a sample video\nvideo_path = '/kaggle/input/sample_video.mp4'\nmodel_path = '/kaggle/working/models/best_model.pth'\noutput_path = '/kaggle/working/output_video.mp4'\n\nprocess_video(video_path, model_path, output_path)\n\"\"\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Final Integration and Testing","metadata":{}},{"cell_type":"code","source":"def run_complete_pipeline():\n    \"\"\"Run the complete deepfake detection pipeline\"\"\"\n    print(\"Deepfake Detection Complete Pipeline\")\n    print(\"Current Date and Time (UTC): 2025-03-04 18:19:19\")\n    print(\"User: AbhiRam162105\")\n    print(\"-\" * 50)\n    \n    # Check if processed data exists\n    try:\n        train_paths = np.load('/kaggle/working/train_paths.npy')\n        train_labels = np.load('/kaggle/working/train_labels.npy')\n        val_paths = np.load('/kaggle/working/val_paths.npy')\n        val_labels = np.load('/kaggle/working/val_labels.npy')\n        \n        print(f\"Found preprocessed data:\")\n        print(f\"  - Training samples: {len(train_paths)}\")\n        print(f\"  - Validation samples: {len(val_paths)}\")\n        \n        # Check for trained model\n        model_path = '/kaggle/working/models/best_model.pth'\n        if os.path.exists(model_path):\n            print(f\"Found trained model at {model_path}\")\n            \n            # Run evaluation\n            print(\"\\nRunning model evaluation...\")\n            metrics, _, _ = evaluate_model(model_path, val_paths, val_labels)\n            \n            # Visualize some predictions\n            print(\"\\nGenerating prediction visualizations...\")\n            visualize_model_predictions(model_path, val_paths, val_labels, num_samples=10)\n            \n            # Visualize activation maps\n            print(\"\\nGenerating activation map visualizations...\")\n            visualize_activation_maps(model_path, val_paths, val_labels, num_samples=5)\n            \n            # Export to ONNX\n            print(\"\\nExporting model to ONNX format...\")\n            onnx_path = convert_to_onnx(model_path)\n            \n            print(\"\\nDeepfake detection pipeline completed successfully!\")\n            print(\"The model can now be deployed using the FastAPI code provided in Cell 15.\")\n            \n        else:\n            print(f\"Trained model not found. Running training pipeline...\")\n            \n            # Train the model\n            model, history = train_deepfake_detector(\n                train_paths, train_labels, val_paths, val_labels, \n                continual_learning=True, backbone='efficientnet-b0')\n            \n            # Save history\n            history_path = '/kaggle/working/models/history.json'\n            with open(history_path, 'w') as f:\n                json.dump(history, f)\n            \n            # Plot training history\n            plot_training_history(history_path)\n            \n            print(\"\\nTraining complete! Use the model for evaluation and deployment.\")\n            \n    except FileNotFoundError:\n        print(\"Processed data not found. Please run the data processing pipeline first (Cells 2-7).\")\n    except Exception as e:\n        print(f\"Error in pipeline: {str(e)}\")\n        import traceback\n        traceback.print_exc()\n\n# Execute the complete pipeline\n# Uncomment the following line to run the entire pipeline\n# run_complete_pipeline()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Continual Learning Demonstration","metadata":{}},{"cell_type":"code","source":"def run_continual_learning_demo():\n    \"\"\"Demonstrate continual learning with new deepfake types\"\"\"\n    print(\"Continual Learning Demonstration\")\n    print(\"Current Date and Time (UTC): 2025-03-04 18:19:19\")\n    print(\"User: AbhiRam162105\")\n    print(\"-\" * 50)\n    \n    try:\n        # Check if model exists\n        model_path = '/kaggle/working/models/best_model.pth'\n        if not os.path.exists(model_path):\n            print(\"Trained base model not found. Please run the training pipeline first.\")\n            return\n            \n        # Load model\n        model = DeepfakeDetector(backbone='efficientnet-b0').to(device)\n        model.load_state_dict(torch.load(model_path))\n        \n        # Initialize continual learning manager\n        cl_manager = ContinualLearningManager(model, device, memory_size=200, strategy='icarl')\n        \n        print(\"Simulating arrival of new deepfake type...\")\n        \n        # In a real-world scenario, you would introduce a new dataset here\n        # For demonstration, we'll create a synthetic new deepfake type by applying \n        # transformations to existing data\n        \n        # Load original data\n        try:\n            val_paths = np.load('/kaggle/working/val_paths.npy')\n            val_labels = np.load('/kaggle/working/val_labels.npy')\n            \n            # Create a synthetic \"new\" deepfake type by selecting some samples\n            print(\"Generating synthetic 'new deepfake' data...\")\n            \n            # Select only fake samples\n            fake_indices = [i for i, label in enumerate(val_labels) if label == 1]\n            fake_paths = [val_paths[i] for i in fake_indices]\n            \n            if len(fake_paths) > 0:\n                # Create output directory\n                new_deepfake_dir = '/kaggle/working/new_deepfake_type'\n                os.makedirs(new_deepfake_dir, exist_ok=True)\n                \n                # Select a subset of fake samples\n                subset_size = min(200, len(fake_paths))\n                selected_paths = fake_paths[:subset_size]\n                \n                # Process these to create a \"new\" deepfake type\n                new_paths = []\n                for i, path in enumerate(tqdm(selected_paths, desc=\"Creating new deepfake type\")):\n                    try:\n                        # Load the image\n                        img = cv2.imread(path)\n                        if img is None:\n                            continue\n                            \n                        # Apply some transformations to simulate a different deepfake type\n                        # (adjust color balance, add slight blur, etc.)\n                        adjusted = img.copy()\n                        \n                        # Adjust color channels (simulating different GAN artifacts)\n                        adjusted[:,:,0] = np.clip(adjusted[:,:,0] * 1.1, 0, 255).astype(np.uint8)  # Blue channel\n                        adjusted[:,:,1] = np.clip(adjusted[:,:,1] * 0.9, 0, 255).astype(np.uint8)  # Green channel\n                        \n                        # Add slight Gaussian blur (simulating different quality)\n                        adjusted = cv2.GaussianBlur(adjusted, (3, 3), 0.5)\n                        \n                        # Add some noise\n                        noise = np.random.normal(0, 5, adjusted.shape).astype(np.uint8)\n                        adjusted = np.clip(adjusted + noise, 0, 255).astype(np.uint8)\n                        \n                        # Save to new location\n                        new_path = os.path.join(new_deepfake_dir, f\"new_deepfake_{i}.jpg\")\n                        cv2.imwrite(new_path, adjusted)\n                        new_paths.append(new_path)\n                        \n                    except Exception as e:\n                        print(f\"Error processing {path}: {str(e)}\")\n                \n                print(f\"Created {len(new_paths)} samples of new deepfake type\")\n                \n                # All are fake (label 1)\n                new_labels = np.ones(len(new_paths), dtype=int)\n                \n                # Initialize memory buffer with samples from original dataset\n                print(\"Initializing memory buffer with original dataset samples...\")\n                train_paths = np.load('/kaggle/working/train_paths.npy')\n                train_labels = np.load('/kaggle/working/train_labels.npy')\n                \n                # Evaluate model before continual learning\n                print(\"\\nEvaluating model BEFORE continual learning:\")\n                before_metrics, _, _ = evaluate_model(model_path, val_paths, val_labels)\n                \n                # Create loader for memory buffer initialization\n                temp_loader, _ = create_spectral_data_loaders(\n                    train_paths[:500], train_labels[:500],  # Use a subset for speed\n                    val_paths[:100], val_labels[:100],\n                    batch_size=32)\n                \n                # Update memory buffer\n                cl_manager.update_memory_buffer(temp_loader, 0)\n                print(f\"Memory buffer initialized with {len(cl_manager.memory_buffer['data'])} examples\")\n                \n                # Run incremental learning\n                print(\"\\nStarting continual learning with new deepfake type...\")\n                model, history = incremental_learning_pipeline(model, new_paths, new_labels, cl_manager)\n                \n                # Evaluate after continual learning\n                print(\"\\nEvaluating model AFTER continual learning:\")\n                after_metrics, _, _ = evaluate_model('/kaggle/working/models/incremental_model_*.pth', val_paths, val_labels)\n                \n                # Evaluate on new deepfake type\n                print(\"\\nEvaluating model on new deepfake type:\")\n                new_metrics, _, _ = evaluate_model('/kaggle/working/models/incremental_model_*.pth', new_paths, new_labels)\n                \n                # Calculate forgetting\n                forgetting_rate = evaluate_catastrophic_forgetting(model, val_paths, val_labels, new_paths, new_labels)\n                \n                # Summary\n                print(\"\\nContinual Learning Summary:\")\n                print(f\"Original dataset performance - Before: {before_metrics['val_acc']:.2f}%, After: {after_metrics['val_acc']:.2f}%\")\n                print(f\"New deepfake type performance: {new_metrics['val_acc']:.2f}%\")\n                print(f\"Change in accuracy: {after_metrics['val_acc'] - before_metrics['val_acc']:.2f}%\")\n                print(f\"Catastrophic forgetting evaluation completed.\")\n                \n            else:\n                print(\"No fake samples found in validation set.\")\n            \n        except FileNotFoundError:\n            print(\"Validation data not found. Please run the data processing pipeline first.\")\n    \n    except Exception as e:\n        print(f\"Error in continual learning demo: {str(e)}\")\n        import traceback\n        traceback.print_exc()\n\n# Uncomment to run the continual learning demo\n# run_continual_learning_demo()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}