{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%pip install mtcnn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:15:37.629169Z","iopub.execute_input":"2025-10-20T13:15:37.629483Z","iopub.status.idle":"2025-10-20T13:15:46.878113Z","shell.execute_reply.started":"2025-10-20T13:15:37.629423Z","shell.execute_reply":"2025-10-20T13:15:46.876889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport json\nimport os, glob\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-20T12:30:09.927341Z","iopub.execute_input":"2025-10-20T12:30:09.927666Z","iopub.status.idle":"2025-10-20T12:30:10.858342Z","shell.execute_reply.started":"2025-10-20T12:30:09.927618Z","shell.execute_reply":"2025-10-20T12:30:10.857513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_VIDEO_DIR = r'/kaggle/input/deepfake-detection-challenge/test_videos'\nTRAIN_VIDEO_DIR = r'/kaggle/input/deepfake-detection-challenge/train_sample_videos'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T12:30:11.498970Z","iopub.execute_input":"2025-10-20T12:30:11.499258Z","iopub.status.idle":"2025-10-20T12:30:11.503375Z","shell.execute_reply.started":"2025-10-20T12:30:11.499215Z","shell.execute_reply":"2025-10-20T12:30:11.502513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_paths = sorted(glob.glob(os.path.join(TEST_VIDEO_DIR, '*.mp4')))\ntrain_paths = sorted(glob.glob(os.path.join(TRAIN_VIDEO_DIR, '*.mp4')))\n\nmetadata_path = os.path.join(TRAIN_VIDEO_DIR, 'metadata.json')\nwith open(metadata_path, 'r') as f:\n    metadata = json.load(f)\n\npd.DataFrame(metadata)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T12:30:12.692208Z","iopub.execute_input":"2025-10-20T12:30:12.692498Z","iopub.status.idle":"2025-10-20T12:30:12.817172Z","shell.execute_reply.started":"2025-10-20T12:30:12.692447Z","shell.execute_reply":"2025-10-20T12:30:12.816090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_iframes(video_path, output_dir, num_frames=20):\n    \"\"\"\n    Extract equidistant frames from a video using cv2.\n    \n    Args:\n        video_path (str): Path to input video file.\n        output_dir (str): Directory where frames will be saved.\n        num_frames (int): Number of frames to extract.\n    \"\"\"\n    # Ensure output directory exists\n    os.makedirs(output_dir, exist_ok=True)\n\n    # Load video\n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        raise ValueError(f\"Error opening video: {video_path}\")\n\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n\n    # If requested frames > available, limit it\n    num_frames = min(num_frames, total_frames)\n\n    # Compute step size\n    step = total_frames // num_frames\n\n    frame_ids = [i * step for i in range(num_frames)]\n\n    for idx, frame_id in enumerate(frame_ids):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, frame_id)\n        ret, frame = cap.read()\n        if ret:\n            frame_filename = os.path.join(output_dir, f\"frame_{idx:04d}.png\")\n            cv2.imwrite(frame_filename, frame)\n        else:\n            print(f\"Warning: could not read frame {frame_id} in {video_path}\")\n\n    cap.release()\n    print(f\"Extracted {len(frame_ids)} frames from {video_path}\")\n    frames = sorted(glob.glob(os.path.join(output_dir, \"frame_*.png\")))\n    return frames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T12:34:12.208003Z","iopub.execute_input":"2025-10-20T12:34:12.208373Z","iopub.status.idle":"2025-10-20T12:34:12.218053Z","shell.execute_reply.started":"2025-10-20T12:34:12.208308Z","shell.execute_reply":"2025-10-20T12:34:12.216978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mtcnn import MTCNN\ndef crop_faces_mtcnn(image_paths, output_dir, min_confidence=0.95, detector=None):\n    \"\"\"\n    Crops the highest-confidence face from each image using MTCNN.\n    Returns a list of saved cropped face image paths.\n    \"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    if detector is None:\n        detector = MTCNN()\n    \n    cropped_files = []\n    for img_path in image_paths:\n        img = cv2.imread(img_path)\n        if img is None:\n            continue\n\n        results = detector.detect_faces(img)\n        if not results:\n            continue\n\n        best_face = max(results, key=lambda x: x['confidence'])\n        conf = best_face['confidence']\n\n        if conf < min_confidence:\n            continue\n\n        x, y, w, h = best_face['box']\n        x, y = max(0, x), max(0, y)\n        face_crop = img[y:y+h, x:x+w]\n\n        out_file = os.path.join(output_dir, os.path.basename(img_path))\n        cv2.imwrite(out_file, face_crop)\n        cropped_files.append(out_file)\n\n    return cropped_files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T12:34:15.382123Z","iopub.execute_input":"2025-10-20T12:34:15.382474Z","iopub.status.idle":"2025-10-20T12:34:15.391743Z","shell.execute_reply.started":"2025-10-20T12:34:15.382408Z","shell.execute_reply":"2025-10-20T12:34:15.390814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_faces_from_video(video_path, work_dir, min_confidence=0.95, max_frames=None, quiet=True):\n    \"\"\"\n    Complete pipeline: extracts I-frames from a video, then crops faces with MTCNN.\n    Returns list of cropped face image paths.\n    \"\"\"\n    base_name = os.path.splitext(os.path.basename(video_path))[0]\n    frames_dir = os.path.join(work_dir, f\"{base_name}_frames\")\n    crops_dir = os.path.join(work_dir, f\"{base_name}_crops\")\n\n    # Step 1: Extract I-frames\n    frames = extract_iframes(video_path, frames_dir)\n\n    # Optional downsampling (if too many frames)\n    if max_frames and len(frames) > max_frames:\n        import random\n        frames = random.sample(frames, max_frames)\n\n    # Step 2: Crop faces\n    detector = MTCNN()\n    cropped_faces = crop_faces_mtcnn(frames, crops_dir, min_confidence=min_confidence, detector=detector)\n\n    return cropped_faces","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T12:35:29.927584Z","iopub.execute_input":"2025-10-20T12:35:29.927927Z","iopub.status.idle":"2025-10-20T12:35:29.935441Z","shell.execute_reply.started":"2025-10-20T12:35:29.927870Z","shell.execute_reply":"2025-10-20T12:35:29.934485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\n\nif __name__ == \"__main__\":\n    video_paths = test_paths\n    for video_path in video_paths:\n        output_root = \"/kaggle/working/processed\"\n\n        cropped_faces = extract_faces_from_video(\n            video_path,\n            work_dir=output_root,\n            min_confidence=0.95,\n        )\n\n        print(f\"Extracted and cropped {len(cropped_faces)} faces from {video_path}\")\n        gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:02:28.889241Z","iopub.execute_input":"2025-10-20T13:02:28.889608Z","iopub.status.idle":"2025-10-20T13:08:04.220871Z","shell.execute_reply.started":"2025-10-20T13:02:28.889553Z","shell.execute_reply":"2025-10-20T13:08:04.219953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -rf /kaggle/working/processed/*_frames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-20T13:09:39.077895Z","iopub.execute_input":"2025-10-20T13:09:39.078228Z","iopub.status.idle":"2025-10-20T13:09:40.357303Z","shell.execute_reply.started":"2025-10-20T13:09:39.078179Z","shell.execute_reply":"2025-10-20T13:09:40.356096Z"}},"outputs":[],"execution_count":null}]}