{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:40.913241Z","iopub.execute_input":"2024-12-03T22:12:40.913548Z","iopub.status.idle":"2024-12-03T22:12:40.917056Z","shell.execute_reply.started":"2024-12-03T22:12:40.913513Z","shell.execute_reply":"2024-12-03T22:12:40.916091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nimport cv2\n\nimport torch.nn.functional as f\nimport torch.optim as optim\n\nimport torchvision\nfrom torchvision import models\nimport torch.optim.lr_scheduler as lr_scheduler\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nfrom tqdm.notebook import tqdm\n\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:41.449986Z","iopub.execute_input":"2024-12-03T22:12:41.450295Z","iopub.status.idle":"2024-12-03T22:12:41.455314Z","shell.execute_reply.started":"2024-12-03T22:12:41.450242Z","shell.execute_reply":"2024-12-03T22:12:41.454621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if torch.cuda.is_available():\n    device = torch.device('cuda')\n    print(\"CUDA is available. Using GPU.\")\nelse:\n    device = torch.device('cpu')\n    print(\"CUDA is not available. Using CPU.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:43.555971Z","iopub.execute_input":"2024-12-03T22:12:43.556267Z","iopub.status.idle":"2024-12-03T22:12:43.619289Z","shell.execute_reply.started":"2024-12-03T22:12:43.556219Z","shell.execute_reply":"2024-12-03T22:12:43.618419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_json(\"/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:43.620771Z","iopub.execute_input":"2024-12-03T22:12:43.621007Z","iopub.status.idle":"2024-12-03T22:12:44.188394Z","shell.execute_reply.started":"2024-12-03T22:12:43.620960Z","shell.execute_reply":"2024-12-03T22:12:44.187706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.T; df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:44.190023Z","iopub.execute_input":"2024-12-03T22:12:44.190407Z","iopub.status.idle":"2024-12-03T22:12:44.211634Z","shell.execute_reply.started":"2024-12-03T22:12:44.190346Z","shell.execute_reply":"2024-12-03T22:12:44.210808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(df[df[\"label\"] == \"FAKE\"]), len(df[df[\"label\"] == \"REAL\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:44.212798Z","iopub.execute_input":"2024-12-03T22:12:44.212984Z","iopub.status.idle":"2024-12-03T22:12:44.220806Z","shell.execute_reply.started":"2024-12-03T22:12:44.212951Z","shell.execute_reply":"2024-12-03T22:12:44.219917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, test_data = train_test_split(\n    df,\n    test_size=0.3,\n    random_state=42,\n    stratify=df[\"label\"]\n)\n\ntest_data, val_data = train_test_split(\n    test_data,\n    test_size=0.5,\n    random_state=42,\n    stratify=test_data[\"label\"]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:44.233938Z","iopub.execute_input":"2024-12-03T22:12:44.234158Z","iopub.status.idle":"2024-12-03T22:12:44.244712Z","shell.execute_reply.started":"2024-12-03T22:12:44.234090Z","shell.execute_reply":"2024-12-03T22:12:44.244055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.shape, val_data.shape, test_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:45.377189Z","iopub.execute_input":"2024-12-03T22:12:45.377472Z","iopub.status.idle":"2024-12-03T22:12:45.382882Z","shell.execute_reply.started":"2024-12-03T22:12:45.377414Z","shell.execute_reply":"2024-12-03T22:12:45.381977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.reset_index(inplace=True)\nval_data.reset_index(inplace=True)\ntest_data.reset_index(inplace=True)\n\ntrain_data = train_data.rename(columns={\"index\": \"videoname\"})\nval_data = val_data.rename(columns={\"index\": \"videoname\"})\ntest_data = test_data.rename(columns={\"index\": \"videoname\"})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:46.326151Z","iopub.execute_input":"2024-12-03T22:12:46.326428Z","iopub.status.idle":"2024-12-03T22:12:46.337267Z","shell.execute_reply.started":"2024-12-03T22:12:46.326387Z","shell.execute_reply":"2024-12-03T22:12:46.336452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:12:47.123900Z","iopub.execute_input":"2024-12-03T22:12:47.124206Z","iopub.status.idle":"2024-12-03T22:12:47.136223Z","shell.execute_reply.started":"2024-12-03T22:12:47.124152Z","shell.execute_reply":"2024-12-03T22:12:47.135385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport json\n\nclass DeepfakeDataset(Dataset):\n    def __init__(self, metadata, video_dir, is_training=True, frames_per_video=5):\n        \"\"\"\n        Args:\n            metadata (str): JSON metadata file containing video labels and info.\n            video_dir (str): Path to the directory containing video files.\n            is_training (bool): Flag to determine if training or validation augmentations are used.\n            frames_per_video (int): Number of frames to sample per video.\n        \"\"\"\n        self.metadata = metadata\n        self.video_dir = video_dir\n        self.frames_per_video = frames_per_video\n        \n        # Collect video paths and labels\n        self.video_files, self.labels = self._retrieve_data()\n        \n        # Define transforms\n        if is_training:\n            self.transform_func = A.Compose([\n                A.Resize(224, 224),\n                A.RandomRotate90(p=0.5),\n                A.HorizontalFlip(p=0.5),\n                A.RandomBrightnessContrast(p=0.2),\n                A.OneOf([\n                    A.GaussNoise(p=1),\n                    A.GaussianBlur(p=1),\n                ], p=0.2),\n                A.Normalize(\n                    mean=[0.485, 0.456, 0.406],\n                    std=[0.229, 0.224, 0.225],\n                ),\n                ToTensorV2()\n            ])\n        else:\n            self.transform_func = A.Compose([\n                A.Resize(224, 224),\n                A.Normalize(\n                    mean=[0.485, 0.456, 0.406],\n                    std=[0.229, 0.224, 0.225],\n                ),\n                ToTensorV2()\n            ])\n    \n    def _retrieve_data(self):\n        video_files = []\n        labels = []\n        \n        for _, row in self.metadata.iterrows():\n            video_name = row['videoname']\n            label = row['label']\n            \n            video_path = os.path.join(self.video_dir, video_name)\n            if os.path.exists(video_path):  # Ensure video file exists\n                video_files.append(video_path)\n                labels.append(1 if label == 'FAKE' else 0)\n    \n        return video_files, labels\n\n    \n    def _extract_frames(self, video_path, num_frames):\n        \"\"\"Extract a fixed number of frames from a video.\"\"\"\n        cap = cv2.VideoCapture(video_path)\n        frames = []\n        total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        frame_indices = np.linspace(0, total_frames - 1, num_frames, dtype=int)\n        \n        for idx in frame_indices:\n            cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n            ret, frame = cap.read()\n            if ret:\n                frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)  # Convert to RGB\n                frames.append(frame)\n        \n        cap.release()\n        return frames\n    \n    def __len__(self):\n        \"\"\"Returns the size of the dataset.\"\"\"\n        return len(self.labels)\n    \n    def __getitem__(self, idx):\n        \"\"\"Fetch a single sample from the dataset.\"\"\"\n        video_path = self.video_files[idx]\n        label = self.labels[idx]\n        \n        # Extract frames and apply transformations\n        frames = self._extract_frames(video_path, self.frames_per_video)\n        transformed_frames = []\n        for frame in frames:\n            transformed = self.transform_func(image=frame)\n            transformed_frames.append(transformed['image'])\n        \n        # Stack frames into a tensor (C, T, H, W) where T is the number of frames\n        video_tensor = torch.stack(transformed_frames)  # Shape: [T, C, H, W]\n        label_tensor = torch.tensor(label, dtype=torch.long)\n        \n        return video_tensor, label_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:13:33.015512Z","iopub.execute_input":"2024-12-03T22:13:33.015792Z","iopub.status.idle":"2024-12-03T22:13:33.030846Z","shell.execute_reply.started":"2024-12-03T22:13:33.015754Z","shell.execute_reply":"2024-12-03T22:13:33.030199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trn_ds = DeepfakeDataset(train_data, '/kaggle/input/deepfake-detection-challenge/train_sample_videos')\nval_ds = DeepfakeDataset(val_data, '/kaggle/input/deepfake-detection-challenge/train_sample_videos')\ntst_ds = DeepfakeDataset(test_data, '/kaggle/input/deepfake-detection-challenge/train_sample_videos')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:16:00.748196Z","iopub.execute_input":"2024-12-03T22:16:00.748490Z","iopub.status.idle":"2024-12-03T22:16:01.605725Z","shell.execute_reply.started":"2024-12-03T22:16:00.748452Z","shell.execute_reply":"2024-12-03T22:16:01.604966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(trn_ds), len(val_ds), len(tst_ds) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:16:23.830286Z","iopub.execute_input":"2024-12-03T22:16:23.830601Z","iopub.status.idle":"2024-12-03T22:16:23.835754Z","shell.execute_reply.started":"2024-12-03T22:16:23.830550Z","shell.execute_reply":"2024-12-03T22:16:23.835035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trn_dl = DataLoader(trn_ds, batch_size=16, shuffle=True, pin_memory=True)\nval_dl = DataLoader(val_ds, batch_size=16, shuffle=True, pin_memory=True)\ntst_dl = DataLoader(tst_ds, batch_size=16, shuffle=True, pin_memory=True)\n\nframes, labels = next(iter(trn_dl))\nprint(frames.shape, labels.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T22:24:01.367755Z","iopub.execute_input":"2024-12-03T22:24:01.368053Z","iopub.status.idle":"2024-12-03T22:24:24.873445Z","shell.execute_reply.started":"2024-12-03T22:24:01.368013Z","shell.execute_reply":"2024-12-03T22:24:24.872262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}