{"metadata":{"colab":{"provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install torchmetrics","metadata":{"id":"qf-o96-vzCZY","outputId":"77b66342-8df1-495f-b668-1c0b246026fd","execution":{"iopub.status.busy":"2024-09-05T18:23:45.623031Z","iopub.execute_input":"2024-09-05T18:23:45.623319Z","iopub.status.idle":"2024-09-05T18:23:59.437805Z","shell.execute_reply.started":"2024-09-05T18:23:45.623287Z","shell.execute_reply":"2024-09-05T18:23:59.436745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install facenet-pytorch","metadata":{"id":"bzyn9oFbttmb","outputId":"7d8b9a8a-5ac7-439d-8ac4-d57408c7668d","execution":{"iopub.status.busy":"2024-09-05T18:23:59.439775Z","iopub.execute_input":"2024-09-05T18:23:59.440105Z","iopub.status.idle":"2024-09-05T18:26:23.121431Z","shell.execute_reply.started":"2024-09-05T18:23:59.440071Z","shell.execute_reply":"2024-09-05T18:26:23.120454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nprint(\"Number of Training videos:\", len(os.listdir('/kaggle/input/deepfake-detection-challenge/train_sample_videos')))\nprint(\"Number of Testing videos:\", len(os.listdir('/kaggle/input/deepfake-detection-challenge/test_videos')))\n# 1 file extra in train_sample_videos is metadata.json","metadata":{"id":"JZkzOXjFAQ7c","outputId":"ba626f15-f0f0-4984-f273-475c74f97642","execution":{"iopub.status.busy":"2024-09-05T18:26:23.122819Z","iopub.execute_input":"2024-09-05T18:26:23.123120Z","iopub.status.idle":"2024-09-05T18:26:23.336127Z","shell.execute_reply.started":"2024-09-05T18:26:23.123087Z","shell.execute_reply":"2024-09-05T18:26:23.335295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\npath = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\nfiles = os.listdir(path)\n\nfor file in files:\n    if file.endswith('.json'):\n        file_path = os.path.join(path, file)\n\nwith open(file_path, 'r') as f:\n    content = f.read()\n\n# converting str to dict\ncontent = json.loads(content)\n\nfake_vid = 0\nreal_vid = 0\nfor vid in content.keys():\n    if content.get(vid)[\"label\"] == \"FAKE\":\n        fake_vid += 1\n    else:\n        real_vid += 1\n\nprint(\"Number of FAKE videos:\", fake_vid)\nprint(\"Number of REAL videos:\", real_vid)","metadata":{"id":"_uFPz7-RtOO3","outputId":"84752ce2-6d79-4b16-f2d8-f2f0ea504e39","execution":{"iopub.status.busy":"2024-09-05T18:26:23.338121Z","iopub.execute_input":"2024-09-05T18:26:23.338409Z","iopub.status.idle":"2024-09-05T18:26:23.350943Z","shell.execute_reply.started":"2024-09-05T18:26:23.338378Z","shell.execute_reply":"2024-09-05T18:26:23.350116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\n\nfake_vid_lst = []\nreal_vid_lst = []\nfor vid in content.keys():\n    if content.get(vid)[\"label\"] == \"FAKE\":\n        fake_vid_lst.append(vid)\n    else:\n        real_vid_lst.append(vid)\n\nall_vid_data = fake_vid_lst[:60] + real_vid_lst[:60]\n# random.shuffle(all_vid_data)\nprint(len(all_vid_data))","metadata":{"id":"AUv-FJTpaDCq","outputId":"28cb5443-f753-47c7-d10c-207ed4528ab3","execution":{"iopub.status.busy":"2024-09-05T18:26:23.352031Z","iopub.execute_input":"2024-09-05T18:26:23.352312Z","iopub.status.idle":"2024-09-05T18:26:23.358749Z","shell.execute_reply.started":"2024-09-05T18:26:23.352281Z","shell.execute_reply":"2024-09-05T18:26:23.357826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(all_vid_data)","metadata":{"id":"-FdxHccLbUkf","outputId":"1512ff09-0585-4ec7-dfa0-1e4004d59015","execution":{"iopub.status.busy":"2024-09-05T18:26:23.360027Z","iopub.execute_input":"2024-09-05T18:26:23.360402Z","iopub.status.idle":"2024-09-05T18:26:23.370102Z","shell.execute_reply.started":"2024-09-05T18:26:23.360359Z","shell.execute_reply":"2024-09-05T18:26:23.369351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\nls = [(vid, content.get(vid).get('label')) for vid in all_vid_data]\nmetadata = pd.DataFrame(ls, columns=['videos', 'label'])\n# metadata.set_index('videos', inplace=True)\n\n\nmetadata['label'] = metadata['label'].map({'FAKE': 0, 'REAL': 1})\nprint(metadata)","metadata":{"id":"Dgb3QWuFthAi","outputId":"bc886ae0-bf7d-4804-cae3-cc77e1e16046","execution":{"iopub.status.busy":"2024-09-05T18:26:23.371215Z","iopub.execute_input":"2024-09-05T18:26:23.371545Z","iopub.status.idle":"2024-09-05T18:26:23.893055Z","shell.execute_reply.started":"2024-09-05T18:26:23.371514Z","shell.execute_reply":"2024-09-05T18:26:23.892109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata[metadata['label'] == 1].shape","metadata":{"id":"_xylUBCxnbCz","outputId":"27b17f36-dfa7-4316-d0d7-da250075b718","execution":{"iopub.status.busy":"2024-09-05T18:26:23.894255Z","iopub.execute_input":"2024-09-05T18:26:23.894731Z","iopub.status.idle":"2024-09-05T18:26:23.902400Z","shell.execute_reply.started":"2024-09-05T18:26:23.894670Z","shell.execute_reply":"2024-09-05T18:26:23.901394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter the first 50 rows where 'label' is 0\nfirst_50_zeros = metadata[metadata['label'] == 0][:50]\n\n# Filter the first 50 rows where 'label' is 1\nfirst_50_ones = metadata[metadata['label'] == 1][:50]\n\n# Concatenate the two DataFrames\ntrain_metadata = pd.concat([first_50_zeros, first_50_ones])\ntrain_metadata.set_index('videos', inplace=True)\n\n# Print the result\nprint(train_metadata.shape)  # This will show the combined DataFrame with 100 rows\n\nlast_10_zeros = metadata[metadata['label'] == 0][50:]\n\nlast_10_ones = metadata[metadata['label'] == 1][50:]\n\ntest_metadata = pd.concat([last_10_zeros, last_10_ones])\n\ntest_metadata.set_index('videos', inplace=True)\n\n# Print the result\nprint(test_metadata.shape)","metadata":{"id":"-828XOSUcVrW","outputId":"572c7d2d-f393-4315-bf64-c9851f7c785d","execution":{"iopub.status.busy":"2024-09-05T18:26:23.903553Z","iopub.execute_input":"2024-09-05T18:26:23.904192Z","iopub.status.idle":"2024-09-05T18:26:23.915887Z","shell.execute_reply.started":"2024-09-05T18:26:23.904159Z","shell.execute_reply":"2024-09-05T18:26:23.914818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset\nimport os\nimport pandas as pd\nfrom torchvision import transforms\nimport cv2\nfrom facenet_pytorch import MTCNN\nimport numpy as np\n\nclass VideoDataset(Dataset):\n    def __init__(self, video_paths, metadata, transform=None, max_frames=200):\n        \"\"\"\n        Args:\n            video_paths (list): List of paths to video files.\n            metadata (pd.DataFrame): DataFrame with metadata indicating fake or original videos.\n            transform (callable, optional): Optional transform to be applied on a sample.\n            max_frames (int): Maximum number of frames to process per video.\n        \"\"\"\n        self.video_paths = video_paths\n        self.metadata = metadata\n        self.transform = transform\n        self.max_frames = max_frames  # Set max frames per video\n\n        # Initialize MTCNN for face detection\n        self.mtcnn = MTCNN(keep_all=True, device='cuda' if torch.cuda.is_available() else 'cpu')\n\n    def __len__(self):\n        return len(self.video_paths)\n\n    def __getitem__(self, idx):\n        video_path = self.video_paths[idx]\n        label = self.metadata.loc[os.path.basename(video_path), 'label']  # Assuming 'label' column exists\n\n        # Load the video\n        cap = cv2.VideoCapture(video_path)\n        face_boxes = []\n        face_labels = []\n\n        frame_count = 0\n\n        while cap.isOpened() and frame_count < self.max_frames:\n            ret, frame = cap.read()\n            if not ret:\n                break\n\n            # Convert frame to RGB (cv2 loads images in BGR format)\n            frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n\n            # Detect faces\n            boxes, _ = self.mtcnn.detect(frame_rgb)\n\n            if boxes is not None:\n                for face in boxes:\n                    face = np.array([round(num) for num in face])\n                    x, y, width, height = face\n                    face_img = frame_rgb[y:y+height, x:x+width]\n\n                    if face_img.shape[0] == 0 or face_img.shape[1] == 0:\n                        continue\n\n                    # Apply transformation\n                    if self.transform:\n                        face_img = self.transform(face_img)\n\n                    face_boxes.append(face_img)\n                    face_labels.append(label)  # 0 for FAKE, 1 for ORIGINAL\n\n            frame_count += 1\n\n        cap.release()\n\n        # Pad or trim the face_boxes and face_labels lists to ensure consistent size\n        face_boxes, face_labels = self.pad_or_trim_frames_labels(face_boxes, face_labels, self.max_frames)\n\n        # Convert face_boxes and face_labels to tensors\n        face_boxes_tensor = torch.stack(face_boxes) if len(face_boxes) > 0 else torch.empty(0)\n        face_labels_tensor = torch.tensor(face_labels, dtype=torch.float32)\n\n        return face_boxes_tensor, face_labels_tensor\n\n    def pad_or_trim_frames_labels(self, frames, labels, max_frames):\n        \"\"\"\n        Pads or trims the list of frames and labels to have a consistent length.\n        Args:\n            frames (list): List of frames (tensors) extracted from the video.\n            labels (list): List of labels corresponding to the frames.\n            max_frames (int): The target number of frames.\n        Returns:\n            tuple: (padded/trimmed frames, padded/trimmed labels)\n        \"\"\"\n        num_frames = len(frames)\n\n        if num_frames < max_frames:\n            # Pad the list of frames with empty tensors\n            padding = [torch.zeros_like(frames[0])] * (max_frames - num_frames)\n            frames.extend(padding)\n\n            # Pad the list of labels with the last label (or you can use a specific value)\n            label_padding = [labels[-1]] * (max_frames - num_frames)\n            labels.extend(label_padding)\n        elif num_frames > max_frames:\n            # Trim the list of frames and labels\n            frames = frames[:max_frames]\n            labels = labels[:max_frames]\n\n        return frames, labels\n","metadata":{"id":"wIIjmgQQtmtE","execution":{"iopub.status.busy":"2024-09-05T18:26:23.919237Z","iopub.execute_input":"2024-09-05T18:26:23.919545Z","iopub.status.idle":"2024-09-05T18:26:27.683904Z","shell.execute_reply.started":"2024-09-05T18:26:23.919514Z","shell.execute_reply":"2024-09-05T18:26:27.683102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader\n\npath_to_folder = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\nvideo_files = list(train_metadata.index)\nvideo_paths = [os.path.join(path_to_folder, video_file) for video_file in video_files]\n\n# Example transformations (if needed)\ntransform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Resize((120, 120)),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\ntrain_dataset = VideoDataset(video_paths=video_paths, metadata=train_metadata, transform=transform)\ntrain_loader = DataLoader(train_dataset, batch_size=4, shuffle=True)\nprint(len(train_loader))","metadata":{"id":"Zc_pEkj7tpdr","outputId":"1d19d01d-c9f8-4a36-eb47-16aa936b3310","execution":{"iopub.status.busy":"2024-09-05T18:26:27.684939Z","iopub.execute_input":"2024-09-05T18:26:27.685307Z","iopub.status.idle":"2024-09-05T18:26:27.952896Z","shell.execute_reply.started":"2024-09-05T18:26:27.685275Z","shell.execute_reply":"2024-09-05T18:26:27.951952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_to_folder = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\nvideo_files = list(test_metadata.index)\nvideo_paths = [os.path.join(path_to_folder, video_file) for video_file in video_files]\n\n# Example transformations (if needed)\ntransform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Resize((120, 120)),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\ntest_dataset = VideoDataset(video_paths=video_paths, metadata=test_metadata, transform=transform)\ntest_loader = DataLoader(test_dataset, batch_size=1, shuffle=True)\nprint(len(test_loader))","metadata":{"id":"dLogLH67GVRq","outputId":"62bcde09-290e-4aee-8e1b-2fcff919975d","execution":{"iopub.status.busy":"2024-09-05T18:26:27.954191Z","iopub.execute_input":"2024-09-05T18:26:27.954867Z","iopub.status.idle":"2024-09-05T18:26:27.981177Z","shell.execute_reply.started":"2024-09-05T18:26:27.954817Z","shell.execute_reply":"2024-09-05T18:26:27.980188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.models as models\n\n# Load a pre-trained VGG16 model\nvgg16 = models.vgg16(weights=True)\n\n# Remove the last fully connected layer (classifier)\nvgg16_features = vgg16.features\n\n# Freeze the pre-trained layers\nfor param in vgg16_features.parameters():\n    param.requires_grad = False","metadata":{"id":"DBzeuvVwxu9N","execution":{"iopub.status.busy":"2024-09-05T18:26:27.982338Z","iopub.execute_input":"2024-09-05T18:26:27.982618Z","iopub.status.idle":"2024-09-05T18:26:33.170710Z","shell.execute_reply.started":"2024-09-05T18:26:27.982587Z","shell.execute_reply":"2024-09-05T18:26:33.169934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\n\nclass CustomVGG16(nn.Module):\n    def __init__(self, num_classes=1):\n        super(CustomVGG16, self).__init__()\n        self.features = vgg16_features  # Use the pre-trained VGG16 features\n        self.avgpool = nn.AdaptiveAvgPool2d((7, 7))  # Ensure correct spatial dimensions\n        self.flatten = nn.Flatten()\n        self.fc1 = nn.Linear(512 * 7 * 7, 2048)  # Custom dense layer 1\n        self.fc2 = nn.Linear(2048, num_classes)  # Custom dense layer 2\n\n    def forward(self, x):\n        x = self.features(x)          # Extract features\n        x = self.avgpool(x)           # Adaptive pooling (optional)\n        x = self.flatten(x)           # Flatten the features\n        x = self.fc1(x)               # Pass through the custom dense layer 1\n        x = nn.ReLU()(x)              # Apply ReLU activation\n        x = self.fc2(x)               # Pass through the custom dense layer 2\n        x = nn.Sigmoid()(x)         # Apply sigmoid activation\n        return x\n","metadata":{"id":"Tuiz4efe6_01","execution":{"iopub.status.busy":"2024-09-05T18:26:33.171822Z","iopub.execute_input":"2024-09-05T18:26:33.172137Z","iopub.status.idle":"2024-09-05T18:26:33.180050Z","shell.execute_reply.started":"2024-09-05T18:26:33.172104Z","shell.execute_reply":"2024-09-05T18:26:33.178946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CustomVGG16()\n\nprint(model)","metadata":{"id":"neyY19p-ENDx","outputId":"1e5ef798-0ffe-4ea3-f948-ccd69118ee24","execution":{"iopub.status.busy":"2024-09-05T18:26:33.181275Z","iopub.execute_input":"2024-09-05T18:26:33.181577Z","iopub.status.idle":"2024-09-05T18:26:33.675044Z","shell.execute_reply.started":"2024-09-05T18:26:33.181544Z","shell.execute_reply":"2024-09-05T18:26:33.674082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchmetrics import Precision, Recall, F1Score\nimport torch\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nprecision_metric = Precision(task='binary', num_classes=1).to(device)\nrecall_metric = Recall(task='binary', num_classes=1).to(device)\nfi_metric = F1Score(task='binary', num_classes=1).to(device)\n\n# Validation step (optional)\ndef validate(val_loader, device, model, criterion):\n  model.eval()\n  val_loss = 0.0\n  correct = 0\n\n  with torch.no_grad():\n      for inputs, labels in val_loader:\n          inputs = inputs.reshape(-1, 3, 120, 120)\n          labels = labels.reshape(-1, 1)\n          inputs, labels = inputs.to(device), labels.to(device)\n          outputs = model(inputs)\n          outputs.reshape(-1,1)\n#           outputs = outputs.squeeze()\n          loss = criterion(outputs, labels.float())\n          val_loss += loss.item() * inputs.size(0)\n\n          # Calculate accuracy\n          preds = torch.round(outputs)\n          correct += (preds == labels).sum().item()\n\n  precision = precision_metric(preds, labels)\n  recall = recall_metric(preds, labels)\n  f1_score = fi_metric(preds, labels)\n\n  val_loss /= len(val_loader.dataset)\n  print(f'Validation Loss: {val_loss:.4f}')\n\n  print(f'Precision: {precision.item():.4f}')\n  print(f'Recall: {recall.item():.4f}')\n  print(f'F1 Score: {f1_score.item():.4f}')\n\n  return precision, recall, f1_score\n\ndef train_loop(num_epochs, train_loader, model, criterion, optimizer, val_loader):\n\n  num_epochs = num_epochs  # Set the number of epochs\n  model.to(device)\n  recall_min = 0.0\n\n  for epoch in range(num_epochs):\n      model.train()\n      running_loss = 0.0\n\n      for inputs, labels in train_loader:\n          inputs = inputs.reshape(-1, 3, 120, 120)\n          labels = labels.reshape(-1, 1)\n          inputs, labels = inputs.to(device), labels.to(device)\n\n          # Zero the parameter gradients\n          optimizer.zero_grad()\n\n          # Forward pass\n          outputs = model(inputs)\n          outputs = outputs.reshape(-1, 1)\n          # outputs = outputs.squeeze()  # Remove dimensions of size 1 for compatibility with BCE loss\n          loss = criterion(outputs, labels.float())  # Ensure labels are float for BCEWithLogitsLoss\n\n          # Backward pass and optimize\n          loss.backward()\n          optimizer.step()\n\n          # Track loss\n          running_loss += loss.item() * inputs.size(0)\n\n      epoch_loss = running_loss / len(train_loader.dataset)\n      print(f'Epoch [{epoch+1}/{num_epochs}], Loss: {epoch_loss:.4f}')\n\n      # Validation step (optional)\n      precision, recall, f1_score = validate(val_loader, device, model, criterion)\n\n      if recall > recall_min:\n        recall_min = recall\n        print(\"Saving model...\")\n        torch.save(model.state_dict(), 'best_model.pth')\n","metadata":{"id":"XVklgShxtNPv","execution":{"iopub.status.busy":"2024-09-05T18:55:37.559948Z","iopub.execute_input":"2024-09-05T18:55:37.560789Z","iopub.status.idle":"2024-09-05T18:55:37.582003Z","shell.execute_reply.started":"2024-09-05T18:55:37.560745Z","shell.execute_reply":"2024-09-05T18:55:37.581088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 3\ncriterion = nn.BCELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n\ntrain_loop(num_epochs, train_loader, model, criterion, optimizer, test_loader)","metadata":{"id":"gvRx1lBrwn7n","outputId":"669a9b51-0608-4938-bd6d-631a25795cd0","execution":{"iopub.status.busy":"2024-09-05T18:36:45.508623Z","iopub.execute_input":"2024-09-05T18:36:45.509346Z","iopub.status.idle":"2024-09-05T18:36:49.435396Z","shell.execute_reply.started":"2024-09-05T18:36:45.509302Z","shell.execute_reply":"2024-09-05T18:36:49.434082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}