{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":13286702,"sourceType":"datasetVersion","datasetId":8420616}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport pandas as pd\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\nimport random\n\n# --- Paths & Parameters ---\nINPUT_DIR = \"/kaggle/input/deepfake-detection-challenge/train_sample_videos/\"\nMETADATA_PATH = os.path.join(INPUT_DIR, 'metadata.json')\nOUTPUT_DIR = \"/kaggle/working/processed_dfdc_images/\"\nFRAMES_PER_VIDEO = 20\nVALIDATION_SPLIT = 0.2\nMAX_VIDEOS_TO_PROCESS = 400\n\nprint(\"Configuration set.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T06:10:27.523599Z","iopub.execute_input":"2025-10-08T06:10:27.524221Z","iopub.status.idle":"2025-10-08T06:10:27.995787Z","shell.execute_reply.started":"2025-10-08T06:10:27.524187Z","shell.execute_reply":"2025-10-08T06:10:27.994898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the metadata using the path defined in Cell 1\nmetadata_df = pd.read_json(METADATA_PATH).T\nprint(f\"Loaded metadata for {len(metadata_df)} videos.\")\n\n# Create the output directories using the path defined in Cell 1\nfor split in ['train', 'validation']:\n    os.makedirs(os.path.join(OUTPUT_DIR, split, 'real'), exist_ok=True)\n    os.makedirs(os.path.join(OUTPUT_DIR, split, 'fake'), exist_ok=True)\nprint(f\"Output folders created at {OUTPUT_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T06:10:27.997090Z","iopub.execute_input":"2025-10-08T06:10:27.997557Z","iopub.status.idle":"2025-10-08T06:10:28.072682Z","shell.execute_reply.started":"2025-10-08T06:10:27.997538Z","shell.execute_reply":"2025-10-08T06:10:28.071889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_frames(video_path, output_folder, max_frames):\n    \"\"\"Extracts a set number of evenly spaced frames from a single video.\"\"\"\n    if not os.path.exists(video_path):\n        return\n        \n    cap = cv2.VideoCapture(video_path)\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if total_frames < max_frames:\n        return\n        \n    video_name = os.path.basename(video_path).split('.')[0]\n    frame_indices = np.linspace(0, total_frames - 1, num=max_frames, dtype=int)\n    \n    for i, frame_index in enumerate(frame_indices):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, frame_index)\n        ret, frame = cap.read()\n        if not ret:\n            continue\n        \n        frame_filename = os.path.join(output_folder, f\"{video_name}_frame_{i}.jpg\")\n        cv2.imwrite(frame_filename, frame)\n        \n    cap.release()\n\nprint(\"Frame extraction function is ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T06:10:28.073401Z","iopub.execute_input":"2025-10-08T06:10:28.073617Z","iopub.status.idle":"2025-10-08T06:10:28.079746Z","shell.execute_reply.started":"2025-10-08T06:10:28.073601Z","shell.execute_reply":"2025-10-08T06:10:28.078994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get a list of videos to process and shuffle them\nvideo_files = list(metadata_df.index)[:MAX_VIDEOS_TO_PROCESS]\nrandom.shuffle(video_files)\n\n# Split the list into training and validation sets\nsplit_index = int(len(video_files) * VALIDATION_SPLIT)\nvalidation_videos = video_files[:split_index]\ntrain_videos = video_files[split_index:]\n\nprint(f\"Splitting data: {len(train_videos)} for training, {len(validation_videos)} for validation.\")\n\n# --- Process Training Videos ---\nfor video_file in tqdm(train_videos, desc=\"Processing TRAIN set\"):\n    video_path = os.path.join(INPUT_DIR, video_file)\n    label = metadata_df.loc[video_file, 'label']\n    \n    if label == 'REAL':\n        output_folder = os.path.join(OUTPUT_DIR, 'train/real')\n    else: # FAKE\n        output_folder = os.path.join(OUTPUT_DIR, 'train/fake')\n    \n    extract_frames(video_path, output_folder, FRAMES_PER_VIDEO)\n\n# --- Process Validation Videos ---\nfor video_file in tqdm(validation_videos, desc=\"Processing VALIDATION set\"):\n    video_path = os.path.join(INPUT_DIR, video_file)\n    label = metadata_df.loc[video_file, 'label']\n    \n    if label == 'REAL':\n        output_folder = os.path.join(OUTPUT_DIR, 'validation/real')\n    else: # FAKE\n        output_folder = os.path.join(OUTPUT_DIR, 'validation/fake')\n        \n    extract_frames(video_path, output_folder, FRAMES_PER_VIDEO)\n\nprint(\"\\n--- Data Preparation Complete! ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T06:10:28.081239Z","iopub.execute_input":"2025-10-08T06:10:28.081489Z"}},"outputs":[],"execution_count":null}]}