{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:45:09.257401Z","iopub.execute_input":"2024-12-28T03:45:09.257717Z","iopub.status.idle":"2024-12-28T03:45:11.758704Z","shell.execute_reply.started":"2024-12-28T03:45:09.257695Z","shell.execute_reply":"2024-12-28T03:45:11.757572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os \nimport json\nimport cv2\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport torch\nimport torch.nn as nn\nfrom torchvision import datasets, transforms, models\nfrom torch.utils.data import DataLoader\nfrom itertools import product\nfrom torch.utils.data import DataLoader, Subset\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score\nfrom torch.nn import Sequential, Linear, MSELoss\nfrom bayes_opt import BayesianOptimization\nfrom torch.utils.data import Subset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:47:42.399570Z","iopub.execute_input":"2024-12-28T03:47:42.399854Z","iopub.status.idle":"2024-12-28T03:47:46.841385Z","shell.execute_reply.started":"2024-12-28T03:47:42.399834Z","shell.execute_reply":"2024-12-28T03:47:46.840775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_path = '/kaggle/input/deepfake-detection-challenge'\ntrain_videos_path = os.path.join(dataset_path, 'train_sample_videos')\ntest_videos_path = os.path.join(dataset_path, 'test_videos')\n\nprint(f\"train samples: {len(os.listdir(os.path.join(dataset_path, train_videos_path)))}\")\nprint(f\"test samples: {len(os.listdir(os.path.join(dataset_path, test_videos_path)))}\")\n\n\ntrain_sample_metadata = pd.read_json('/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:45:26.322001Z","iopub.execute_input":"2024-12-28T03:45:26.322318Z","iopub.status.idle":"2024-12-28T03:45:26.411680Z","shell.execute_reply.started":"2024-12-28T03:45:26.322291Z","shell.execute_reply":"2024-12-28T03:45:26.410832Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"****Extracting frames****","metadata":{}},{"cell_type":"code","source":"output_dir = '/kaggle/working/frames'\nos.makedirs(output_dir, exist_ok=True)\n\nreal_folder = os.path.join(output_dir, 'REAL')\nfake_folder = os.path.join(output_dir, 'Fake')\nos.makedirs(real_folder, exist_ok = True)\nos.makedirs(fake_folder, exist_ok = True)\n\n# frame_rate: The frequency of frame extraction, default is 1(we may need to increase it)\ndef extract_frames(video_path, extracted_frames, frame_rate = 30):\n    \n    cap = cv2.VideoCapture(video_path)\n    count = 0\n    while cap.isOpened():\n        retrn, frame = cap.read()\n        if not retrn:\n            break\n        if int(cap.get(cv2.CAP_PROP_POS_FRAMES)) % frame_rate == 0:\n            frame_path = os.path.join(extracted_frames, f\"{os.path.basename(video_path)}_frame_{count}.jpg\")\n            frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n            cv2.imwrite(frame_path, frame) # saves the frame as an image\n            count += 1\n            # fig = plt.figure(figsize=(4, 4))\n            # ax = fig.add_subplot(111)  \n            # ax.imshow(frame)\n            # plt.show()\n            # count += 1\n            \n    cap.release() # closes the video file, cleans up memory buffer\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:47:09.993653Z","iopub.execute_input":"2024-12-28T03:47:09.993947Z","iopub.status.idle":"2024-12-28T03:47:10.000205Z","shell.execute_reply.started":"2024-12-28T03:47:09.993926Z","shell.execute_reply":"2024-12-28T03:47:09.999474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Extract frames for each video\nfor video_name, data in train_sample_metadata.iterrows():\n    label = data['label']  # Access the label column\n    video_path = os.path.join(train_videos_path, video_name)  # Construct the full video path\n    extracted_frames = real_folder if label == 'REAL' else fake_folder\n    \n    # Extract frames from the video\n    extract_frames(video_path, extracted_frames)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:47:51.860821Z","iopub.execute_input":"2024-12-28T03:47:51.861221Z","iopub.status.idle":"2024-12-28T03:57:08.912568Z","shell.execute_reply.started":"2024-12-28T03:47:51.861200Z","shell.execute_reply":"2024-12-28T03:57:08.911827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimport os\n\n# Extract video names and labels from the DataFrame\nvideo_names = train_sample_metadata.index.tolist()  # Video names are the index of the DataFrame\nlabels = [1 if train_sample_metadata.loc[v, 'label'] == 'FAKE' else 0 for v in video_names]\n\n# Perform 3-fold cross-validation\nskf = StratifiedKFold(n_splits=3, shuffle=True, random_state=42)\nfolds = []\n\nfor fold, (train_idx, test_idx) in enumerate(skf.split(video_names, labels)):\n    train_videos = [video_names[i] for i in train_idx]\n    test_videos = [video_names[i] for i in test_idx]\n    folds.append({'train': train_videos, 'test': test_videos})\n    print(f\"Fold {fold + 1} - Train: {len(train_videos)}, Test: {len(test_videos)}\")\n\n# Save fold information\nfold_dir = '/kaggle/working/folds_data'\nos.makedirs(fold_dir, exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:58:46.776208Z","iopub.execute_input":"2024-12-28T03:58:46.776499Z","iopub.status.idle":"2024-12-28T03:58:46.791783Z","shell.execute_reply.started":"2024-12-28T03:58:46.776477Z","shell.execute_reply":"2024-12-28T03:58:46.790749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\n# Organize data into fold directories and copy multiple frames per video\nnum_frames_to_use = 10\n\nfor fold_idx, fold in enumerate(folds):\n    fold_output = os.path.join(fold_dir, f'fold_{fold_idx+1}')\n    train_output = os.path.join(fold_output, 'train')\n    test_output = os.path.join(fold_output, 'test')\n\n    os.makedirs(train_output, exist_ok=True)\n    os.makedirs(test_output, exist_ok=True)\n\n    # Copy multiple frames for training videos\n    for video_name in fold['train']:\n        label = train_sample_metadata.loc[video_name, 'label']  # Use .loc to access the 'label' column\n        src_dir = real_folder if label == 'REAL' else fake_folder\n        for frame_idx in range(num_frames_to_use):\n            src_path = os.path.join(src_dir, f\"{video_name}_frame_{frame_idx}.jpg\")\n            if os.path.exists(src_path):  # Ensure the frame exists\n                dst_dir = os.path.join(train_output, label)\n                os.makedirs(dst_dir, exist_ok=True)\n                shutil.copy(src_path, dst_dir)\n\n    # Copy multiple frames for test videos\n    for video_name in fold['test']:\n        label = train_sample_metadata.loc[video_name, 'label']  # Use .loc to access the 'label' column\n        src_dir = real_folder if label == 'REAL' else fake_folder\n        for frame_idx in range(num_frames_to_use):\n            src_path = os.path.join(src_dir, f\"{video_name}_frame_{frame_idx}.jpg\")\n            if os.path.exists(src_path):  # Ensure the frame exists\n                dst_dir = os.path.join(test_output, label)\n                os.makedirs(dst_dir, exist_ok=True)\n                shutil.copy(src_path, dst_dir)\n\nprint(\"Data successfully organized into folds.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T03:58:58.573151Z","iopub.execute_input":"2024-12-28T03:58:58.573422Z","iopub.status.idle":"2024-12-28T03:59:02.918046Z","shell.execute_reply.started":"2024-12-28T03:58:58.573403Z","shell.execute_reply":"2024-12-28T03:59:02.916991Z"}},"outputs":[],"execution_count":null}]}