{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-04T21:47:34.344402Z","iopub.execute_input":"2024-12-04T21:47:34.34493Z","iopub.status.idle":"2024-12-04T21:47:36.212567Z","shell.execute_reply.started":"2024-12-04T21:47:34.344894Z","shell.execute_reply":"2024-12-04T21:47:36.211295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\n\n# مسار ملف metadata\nmetadata_path = '/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json'\n\n# قراءة بيانات metadata\nwith open(metadata_path, 'r') as f:\n    metadata = json.load(f)\n\nmetadata_df = pd.DataFrame.from_dict(metadata, orient='index').reset_index()\nmetadata_df.columns = ['filename', 'label', 'split', 'original']\n\n# إضافة عمود للتقسيم\nskf = StratifiedKFold(n_splits=3, shuffle=True, random_state=42)\n\n# إضافة fold لكل فيديو\nmetadata_df['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(skf.split(metadata_df, metadata_df['label'])):\n    metadata_df.loc[val_idx, 'fold'] = fold\n\n# التأكد من النتيجة\nprint(metadata_df.head())\n\n# حفظ الملف المعدل\nmetadata_df.to_csv('updated_metadata.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T21:48:07.028521Z","iopub.execute_input":"2024-12-04T21:48:07.028867Z","iopub.status.idle":"2024-12-04T21:48:07.636025Z","shell.execute_reply.started":"2024-12-04T21:48:07.028838Z","shell.execute_reply":"2024-12-04T21:48:07.63502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport random\nfrom torch.utils.data import random_split, DataLoader, Subset,Dataset\nfrom torchvision import transforms, models\nfrom sklearn.model_selection import KFold, train_test_split\nfrom PIL import Image\nfrom torch import optim\nimport torch\nimport os\nimport numpy as np\nimport torch.nn as nn\nimport random\nimport shutil\n\nimport cv2\n\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score, precision_score, recall_score, f1_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:31:15.769632Z","iopub.execute_input":"2024-12-05T04:31:15.769956Z","iopub.status.idle":"2024-12-05T04:31:15.776732Z","shell.execute_reply.started":"2024-12-05T04:31:15.769907Z","shell.execute_reply":"2024-12-05T04:31:15.775409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport random\nimport numpy as np\n\n# مسارات الفيديوهات والمجلدات\nvideo_dir = \"/kaggle/input/deepfake-detection-challenge/train_sample_videos\"\nmetadata_path = \"/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json\"\noutput_dir = \"./frames\"\n\n# قراءة بيانات metadata\nwith open(metadata_path, \"r\") as f:\n    metadata = json.load(f)\n\n# divide data into real and fake videos\nreal_videos = [video for video, info in metadata.items() if info['label'] == 'REAL']\nfake_videos = [video for video, info in metadata.items() if info['label'] == 'FAKE']\n\n# take random 50 video from the dataset\nselected_real_videos = random.sample(real_videos, 20)\nselected_fake_videos = random.sample(fake_videos, 20)\n\n#selected videos\nselected_videos = selected_real_videos + selected_fake_videos\n\n# create folders for real and fake videos frames\nos.makedirs(f\"{output_dir}/REAL\", exist_ok=True)\nos.makedirs(f\"{output_dir}/FAKE\", exist_ok=True)\n\n# Extract frames for only selected videos\nfor video_name in selected_videos:\n    video_path = os.path.join(video_dir, video_name)\n    label = metadata[video_name]['label']  # REAL or FAKE\n    output_folder = os.path.join(output_dir, label, os.path.splitext(video_name)[0])\n    \n    os.makedirs(output_folder, exist_ok=True)\n\n    # فتح الفيديو باستخدام OpenCV\n    cap = cv2.VideoCapture(video_path)\n    frame_count = 0\n    frames = []\n\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n        frames.append(frame)  # إضافة الإطار لقائمة الإطارات\n        frame_count += 1\n\n    cap.release()\n\n    # اختيار 100 إطار عشوائي إذا كانت الإطارات أكثر من 100\n    if frame_count > 100:\n        selected_indices = np.linspace(0, frame_count - 1, 100, dtype=int)\n        frames = [frames[i] for i in selected_indices]\n    \n    # حفظ الإطارات المحددة\n    for idx, frame in enumerate(frames):\n        frame_filename = os.path.join(output_folder, f\"frame_{idx:04d}.jpg\")\n        cv2.imwrite(frame_filename, frame)\n    \n    print(f\"{frame_count} frames extracted from: {video_name}\")\n\n# display selected videos\nprint(\"selected real videos:\", selected_real_videos)\nprint(\"selected fake videos:\", selected_fake_videos)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:31:19.944289Z","iopub.execute_input":"2024-12-05T04:31:19.944644Z","iopub.status.idle":"2024-12-05T04:35:24.761439Z","shell.execute_reply.started":"2024-12-05T04:31:19.944593Z","shell.execute_reply":"2024-12-05T04:35:24.760427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport torch\nfrom torch.utils.data import Dataset\nimport numpy as np\n\nclass FrameDataset(Dataset):\n    def __init__(self, frames_dir, transform=None, num_frames=100):\n        self.frames_dir = frames_dir\n        self.video_folders = sorted(os.listdir(frames_dir))  # Subdirectories for each video\n        self.transform = transform\n        self.num_frames = num_frames\n\n        # التحقق من وجود المجلد\n        if not os.path.exists(self.frames_dir):\n            raise FileNotFoundError(f\"Frames directory {self.frames_dir} does not exist.\")\n\n    def __len__(self):\n        return len(self.video_folders)\n\n    def __getitem__(self, idx):\n        video_folder = os.path.join(self.frames_dir, self.video_folders[idx])\n        frame_files = sorted(os.listdir(video_folder))\n\n        # التحقق من وجود إطارات\n        if not frame_files:\n            raise ValueError(f\"No frames found in {video_folder}\")\n\n        frames = []\n\n        # Apply transformation to all frames\n        for frame_file in frame_files:\n            frame_path = os.path.join(video_folder, frame_file)\n            image = Image.open(frame_path).convert(\"RGB\")\n            if self.transform:\n                image = self.transform(image)\n            frames.append(image)\n\n        # Truncate or sample to the desired number of frames\n        if len(frames) > self.num_frames:\n            indices = np.linspace(0, len(frames) - 1, self.num_frames, dtype=int)\n            frames = [frames[i] for i in indices]\n        else:\n            padding = self.num_frames - len(frames)\n            padding_frames = [torch.zeros_like(frames[0]) for _ in range(padding)]\n            frames.extend(padding_frames)\n\n        # Stack frames into a tensor\n        frames = torch.stack(frames)\n\n        # Return video tensor and label\n        label_map = {'REAL': 1, 'FAKE': 0}\n        label = label_map[os.path.basename(os.path.dirname(video_folder)).upper()]\n        return frames, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:35:56.906254Z","iopub.execute_input":"2024-12-05T04:35:56.906849Z","iopub.status.idle":"2024-12-05T04:35:56.922603Z","shell.execute_reply.started":"2024-12-05T04:35:56.906574Z","shell.execute_reply":"2024-12-05T04:35:56.921548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define preprocessing transformations\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor()\n])\n\n# Load datasets\nfake_dataset = FrameDataset('/kaggle/working/frames/FAKE', transform=transform, num_frames=100)\nreal_dataset = FrameDataset('/kaggle/working/frames/REAL', transform=transform, num_frames=100)\n\n# Combine datasets and create DataLoader\ncombined_dataset = fake_dataset + real_dataset\ndataloader = DataLoader(combined_dataset, batch_size=4, shuffle=True)\n\n# Check the output\nfor frames, labels in dataloader:\n    print(frames.shape)  # Shape will be [batch_size, num_frames, 3, 224, 224]\n    print(labels)\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:36:01.461808Z","iopub.execute_input":"2024-12-05T04:36:01.462171Z","iopub.status.idle":"2024-12-05T04:36:18.073996Z","shell.execute_reply.started":"2024-12-05T04:36:01.462109Z","shell.execute_reply":"2024-12-05T04:36:18.073074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for frames, labels in dataloader:\n    print(f\"Batch shape: {frames.shape}\")  \n    print(f\"Labels: {labels}\")            \n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:36:18.076779Z","iopub.execute_input":"2024-12-05T04:36:18.077126Z","iopub.status.idle":"2024-12-05T04:36:35.15601Z","shell.execute_reply.started":"2024-12-05T04:36:18.077063Z","shell.execute_reply":"2024-12-05T04:36:35.154935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine fake and real datasets\ncombined_dataset = fake_dataset + real_dataset\nlabels = [0] * len(fake_dataset) + [1] * len(real_dataset)  # (0 for fake, 1 for real)\n\n# Split the combined dataset\ntrain_val_idx, test_idx = train_test_split(\n    range(len(combined_dataset)), test_size=0.2, stratify=labels, random_state=42\n)\n\ntrain_val_dataset = Subset(combined_dataset, train_val_idx)\ntest_dataset = Subset(combined_dataset, test_idx)\n\n# Create DataLoader\ntest_loader = DataLoader(test_dataset, batch_size=4, shuffle=False)\n\nprint(f\"Total dataset size: {len(combined_dataset)}\")\nprint(f\"Train+Validation dataset size: {len(train_val_dataset)}\")\nprint(f\"Test dataset size: {len(test_dataset)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:36:35.170971Z","iopub.execute_input":"2024-12-05T04:36:35.171319Z","iopub.status.idle":"2024-12-05T04:36:35.18339Z","shell.execute_reply.started":"2024-12-05T04:36:35.171259Z","shell.execute_reply":"2024-12-05T04:36:35.182029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cpu\")\n\ndef cross_validation(model, dataset, criterion, optimizer, device, epochs=5, n_splits=3):\n\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    fold_metrics = {\"accuracy\": [], \"precision\": [], \"recall\": [], \"f1_score\": []}\n\n    for fold, (train_idx, val_idx) in enumerate(kf.split(dataset)):\n        print(f\"\\n--- Fold {fold + 1} ---\")\n\n        # Split dataset into\n        train_subset = Subset(dataset, train_idx)\n        val_subset = Subset(dataset, val_idx)\n\n        train_loader = DataLoader(train_subset, batch_size=4, shuffle=True)\n        val_loader = DataLoader(val_subset, batch_size=4, shuffle=False)\n\n        # Train the model for this fold\n        train_model(model, train_loader, val_loader, criterion, optimizer, epochs, device)\n\n        # Evaluate the model for this fold\n        val_labels, val_preds = evaluate_model(model, val_loader, device)\n\n        # Compute metrics for this fold\n        accuracy = accuracy_score(val_labels, val_preds)\n        precision = precision_score(val_labels, val_preds, average=\"weighted\")\n        recall = recall_score(val_labels, val_preds, average=\"weighted\")\n        f1 = f1_score(val_labels, val_preds, average=\"weighted\")\n\n        fold_metrics[\"accuracy\"].append(accuracy)\n        fold_metrics[\"precision\"].append(precision)\n        fold_metrics[\"recall\"].append(recall)\n        fold_metrics[\"f1_score\"].append(f1)\n\n        print(f\"Fold {fold + 1} Metrics:\")\n        print(f\"  Accuracy: {accuracy:.4f}, Precision: {precision:.4f}, \"\n              f\"Recall: {recall:.4f}, F1-Score: {f1:.4f}\")\n\n    # Compute avg metrics across all folds\n    print(\"\\nAverage Metrics Across Folds:\")\n    print(f\"  Accuracy: {np.mean(fold_metrics['accuracy']):.4f}, \"\n          f\"Precision: {np.mean(fold_metrics['precision']):.4f}, \"\n          f\"Recall: {np.mean(fold_metrics['recall']):.4f}, \"\n          f\"F1-Score: {np.mean(fold_metrics['f1_score']):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:36:39.859201Z","iopub.execute_input":"2024-12-05T04:36:39.859716Z","iopub.status.idle":"2024-12-05T04:36:39.87402Z","shell.execute_reply.started":"2024-12-05T04:36:39.859631Z","shell.execute_reply":"2024-12-05T04:36:39.873029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define model (AlexNet)\nalexnet = models.alexnet(pretrained=True)\nalexnet.classifier[6] = nn.Sequential(\n    nn.Dropout(0.5),\n    nn.Linear(alexnet.classifier[6].in_features, 2)\n)\n\nalexnet = alexnet.to(device)\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(alexnet.parameters(), lr=0.001)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T21:54:39.667381Z","iopub.execute_input":"2024-12-04T21:54:39.668491Z","iopub.status.idle":"2024-12-04T21:54:41.578343Z","shell.execute_reply.started":"2024-12-04T21:54:39.668447Z","shell.execute_reply":"2024-12-04T21:54:41.577137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, criterion, optimizer, epochs, device):\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n\n        # Training loop\n        for videos, labels in train_loader:\n            labels = labels.to(device)\n\n            optimizer.zero_grad()\n\n            # Process each frame in the video\n            batch_predictions = []\n            for video_frames in videos:\n                video_frames = video_frames.to(device)\n\n                # Process frames one by one\n                frame_outputs = model(video_frames)\n                video_prediction = torch.mean(frame_outputs, dim=0)\n                batch_predictions.append(video_prediction)\n\n            batch_predictions = torch.stack(batch_predictions)\n            loss = criterion(batch_predictions, labels)\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item() * videos.size(0)\n\n        # Validation loop\n        val_loss = 0.0\n        val_corrects = 0\n        model.eval()\n        with torch.no_grad():\n            for videos, labels in val_loader:\n                labels = labels.to(device)\n\n                batch_predictions = []\n                for video_frames in videos:\n                    video_frames = video_frames.to(device)\n                    frame_outputs = model(video_frames)\n                    video_prediction = torch.mean(frame_outputs, dim=0)\n                    batch_predictions.append(video_prediction)\n\n                batch_predictions = torch.stack(batch_predictions)\n                loss = criterion(batch_predictions, labels)\n                val_loss += loss.item() * videos.size(0)\n                preds = torch.argmax(batch_predictions, dim=1)\n                val_corrects += torch.sum(preds == labels.data)\n\n        # Print epoch performance\n        print(f\"Epoch {epoch + 1}/{epochs}, \"\n              f\"Train Loss: {train_loss / len(train_loader.dataset):.4f}, \"\n              f\"Val Loss: {val_loss / len(val_loader.dataset):.4f}, \"\n              f\"Val Accuracy: {val_corrects.double() / len(val_loader.dataset):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T21:54:59.216677Z","iopub.execute_input":"2024-12-04T21:54:59.217693Z","iopub.status.idle":"2024-12-04T21:54:59.23068Z","shell.execute_reply.started":"2024-12-04T21:54:59.21765Z","shell.execute_reply":"2024-12-04T21:54:59.22948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## ResNet ##\ndef evaluate_model(model, data_loader, device):\n    model.eval()\n    all_preds = []\n    all_labels = []\n\n    with torch.no_grad():\n        for videos, labels in data_loader:\n            videos, labels = videos.to(device), labels.to(device)\n\n            # Process each video in the batch\n            batch_predictions = []\n            for video_frames in videos:\n                video_frames = video_frames.to(device)\n                frame_outputs = model(video_frames)\n                video_prediction = torch.mean(frame_outputs, dim=0)\n                batch_predictions.append(video_prediction)\n\n            batch_predictions = torch.stack(batch_predictions)\n            preds = torch.argmax(batch_predictions, dim=1)\n            all_preds.extend(preds.cpu().numpy())\n            all_labels.extend(labels.cpu().numpy())\n\n    return np.array(all_labels), np.array(all_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T21:55:21.351914Z","iopub.execute_input":"2024-12-04T21:55:21.35237Z","iopub.status.idle":"2024-12-04T21:55:21.35968Z","shell.execute_reply.started":"2024-12-04T21:55:21.352323Z","shell.execute_reply":"2024-12-04T21:55:21.358567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cross_validation(alexnet, train_val_dataset, criterion, optimizer, device, epochs=3, n_splits=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T21:56:30.807198Z","iopub.execute_input":"2024-12-04T21:56:30.807574Z","iopub.status.idle":"2024-12-04T22:32:13.300256Z","shell.execute_reply.started":"2024-12-04T21:56:30.807544Z","shell.execute_reply":"2024-12-04T22:32:13.299133Z"}},"outputs":[],"execution_count":null}]}