{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[]},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"id":"HACCQkBbo19T","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# Deepfake Detection with Data Generator for Kaggle Notebooks\n# This code provides a complete, runnable solution for training on the large\n# DFDC dataset by using a custom data generator to handle memory constraints.\n# ==============================================================================\n\nimport os\nimport cv2\nimport json\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, TimeDistributed, Dense, LSTM, Dropout, Flatten\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications import Xception\nfrom tensorflow.keras.utils import Sequence\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score, roc_auc_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:20.855347Z","iopub.execute_input":"2025-08-26T19:47:20.855695Z","iopub.status.idle":"2025-08-26T19:47:20.860678Z","shell.execute_reply.started":"2025-08-26T19:47:20.855634Z","shell.execute_reply":"2025-08-26T19:47:20.859903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# 1. Project Configuration and Hyperparameters\n# ==============================================================================\n# The size of the face images after cropping and resizing\nIMG_SIZE = 224\n# The number of frames to sample from each video for the sequence\nMAX_SEQ_LENGTH = 12\n# The training batch size. Adjust based on your GPU memory.\nBATCH_SIZE = 8\n# Number of training epochs\nEPOCHS = 10\n\n# --- Kaggle-specific dataset path ---\nDATA_DIR = \"/kaggle/input/deepfake-detection-challenge\"\n\n# Path to the pre-trained Haar Cascade face detector XML file\nHAAR_CASCADE_PATH = cv2.data.haarcascades + 'haarcascade_frontalface_default.xml'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:20.862278Z","iopub.execute_input":"2025-08-26T19:47:20.862573Z","iopub.status.idle":"2025-08-26T19:47:20.873708Z","shell.execute_reply.started":"2025-08-26T19:47:20.862532Z","shell.execute_reply":"2025-08-26T19:47:20.873073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# 2. Data Pre-processing Pipeline\n# ==============================================================================\n\ndef extract_faces_from_video(video_path):\n    \"\"\"\n    Extracts a fixed number of face-cropped frames from a video.\n    This function simulates the pre-processing workflow for deepfake detection.\n    \n    Args:\n        video_path (str): The path to the video file.\n    \n    Returns:\n        np.array: A 4D numpy array of pre-processed face images (frames).\n    \"\"\"\n    frames = []\n    # Load the pre-trained Haar Cascade classifier for face detection\n    face_cascade = cv2.CascadeClassifier(HAAR_CASCADE_PATH)\n    \n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        # Handle cases where the video file cannot be opened\n        return np.zeros((MAX_SEQ_LENGTH, IMG_SIZE, IMG_SIZE, 3))\n\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    \n    # Calculate an interval to evenly sample frames\n    frame_interval = max(1, total_frames // MAX_SEQ_LENGTH)\n    frame_count = 0\n    \n    while len(frames) < MAX_SEQ_LENGTH:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, frame_count)\n        ret, frame = cap.read()\n        \n        if not ret:\n            break\n            \n        # Convert frame to grayscale for faster face detection\n        gray_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n        \n        # Detect faces using the Haar Cascade classifier\n        # minSize=(40,40) helps filter out very small detections\n        faces = face_cascade.detectMultiScale(gray_frame, scaleFactor=1.1, minNeighbors=5, minSize=(40, 40))\n        \n        if len(faces) > 0:\n            # Assume only one face per video for simplicity; take the largest one.\n            (x, y, w, h) = sorted(faces, key=lambda f: f[2] * f[3], reverse=True)[0]\n            \n            # Crop the detected face region\n            face_img = frame[y:y+h, x:x+w]\n            \n            # Resize and normalize the face image\n            resized_face = cv2.resize(face_img, (IMG_SIZE, IMG_SIZE))\n            normalized_face = resized_face / 255.0\n            \n            frames.append(normalized_face)\n            \n        frame_count += frame_interval\n        \n  # ...\n    cap.release()\n    \n    # --- FIX START ---\n    # Moved the 'if not frames:' check BEFORE converting to a numpy array\n    # to avoid the ValueError.\n    if not frames:\n        return np.zeros((MAX_SEQ_LENGTH, IMG_SIZE, IMG_SIZE, 3))\n    \n    frames = np.array(frames)\n    # --- FIX END ---\n    \n    # Pad sequences with zeros if fewer than MAX_SEQ_LENGTH faces were detected\n    if len(frames) < MAX_SEQ_LENGTH:\n        padding_needed = MAX_SEQ_LENGTH - len(frames)\n        padding_shape = (padding_needed, IMG_SIZE, IMG_SIZE, 3)\n        padded_frames = np.zeros(padding_shape)\n        frames = np.vstack((frames, padded_frames))\n        \n    return frames\n\n\ndef load_dataset_from_dir(data_dir):\n    \"\"\"\n    Loads video paths and labels from the DFDC dataset structure.\n    \n    Args:\n        data_dir (str): The root directory of the dataset.\n        \n    Returns:\n        tuple: A tuple containing lists of (video_paths, labels).\n    \"\"\"\n    print(\"Loading dataset metadata from disk...\")\n    video_paths = []\n    labels = []\n    \n    # DFDC dataset is organized into 'train_sample_videos', 'test_videos', etc.\n    # The training data is further split into parts (e.g., 'part_00', 'part_01')\n    # in the full competition dataset. We'll iterate through these parts.\n    \n    # For this example, we will just use the 'train_sample_videos' part\n    # which is often available by default for quick testing.\n    # To use the full dataset, you would need to iterate through all 'part_xx' folders.\n    \n    dataset_part_dir = os.path.join(data_dir, 'train_sample_videos')\n    metadata_path = os.path.join(dataset_part_dir, 'metadata.json')\n    \n    if os.path.exists(metadata_path):\n        with open(metadata_path, 'r') as f:\n            metadata = json.load(f)\n            \n        for video_name, video_info in metadata.items():\n            # We only care about videos that are marked FAKE or REAL.\n            if video_info['label'] in ['FAKE', 'REAL']:\n                video_path = os.path.join(dataset_part_dir, video_name)\n                if os.path.exists(video_path):\n                    video_paths.append(video_path)\n                    labels.append(1 if video_info['label'] == 'FAKE' else 0)\n    else:\n        print(f\"Error: metadata.json not found at {metadata_path}\")\n    \n    return video_paths, np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:20.875652Z","iopub.execute_input":"2025-08-26T19:47:20.875942Z","iopub.status.idle":"2025-08-26T19:47:20.893048Z","shell.execute_reply.started":"2025-08-26T19:47:20.875887Z","shell.execute_reply":"2025-08-26T19:47:20.892221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# 3. Model Architecture (CNN-LSTM Hybrid)\n# ==============================================================================\n\ndef create_model():\n    \"\"\"\n    Builds and compiles the hybrid CNN-LSTM model for deepfake detection.\n    \n    Returns:\n        tf.keras.models.Model: The compiled Keras model.\n    \"\"\"\n    # Create the CNN backbone using a pre-trained Xception model.\n    cnn_backbone = Xception(\n        weights=\"imagenet\", # Use pre-trained weights from ImageNet\n        include_top=False,  # Exclude the final classification layer\n        input_shape=(IMG_SIZE, IMG_SIZE, 3)\n    )\n    cnn_backbone.trainable = True # Set the backbone to be trainable for fine-tuning\n    \n    # Define the model input as a sequence of video frames\n    video_input = Input(shape=(MAX_SEQ_LENGTH, IMG_SIZE, IMG_SIZE, 3))\n    \n    # Use TimeDistributed to apply the CNN to each frame in the sequence\n    cnn_features = TimeDistributed(cnn_backbone)(video_input)\n    \n    # Flatten the features from the CNN for the LSTM input\n    cnn_features = TimeDistributed(Flatten())(cnn_features)\n    \n    # Add the LSTM layer for temporal analysis\n    lstm_features = LSTM(128)(cnn_features)\n    \n    # Add a Dropout layer for regularization to prevent overfitting\n    lstm_features = Dropout(0.5)(lstm_features)\n    \n    # The final output layer for binary classification (Real vs. Fake)\n    output = Dense(1, activation='sigmoid')(lstm_features)\n    \n    # Construct the full model from the input and output layers\n    model = Model(inputs=video_input, outputs=output)\n    \n    # Compile the model with an optimizer, loss function, and metrics\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n        loss='binary_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    model.summary()\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:20.973432Z","iopub.execute_input":"2025-08-26T19:47:20.973735Z","iopub.status.idle":"2025-08-26T19:47:20.981190Z","shell.execute_reply.started":"2025-08-26T19:47:20.973683Z","shell.execute_reply":"2025-08-26T19:47:20.980352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# 4. Data Generator for Training on Large Datasets\n# ==============================================================================\n\nclass DataGenerator(Sequence):\n    \"\"\"\n    Keras Data Generator for efficiently loading and pre-processing videos\n    in batches, preventing memory errors on large datasets.\n    \"\"\"\n    def __init__(self, video_paths, labels, batch_size=BATCH_SIZE, shuffle=True):\n        self.video_paths = video_paths\n        self.labels = labels\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Returns the number of batches per epoch.\"\"\"\n        return int(np.floor(len(self.video_paths) / self.batch_size))\n\n    def __getitem__(self, index):\n        \"\"\"Generates one batch of data.\"\"\"\n        # Get the batch's indices\n        indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        \n        # Get the video paths and labels for this batch\n        batch_paths = [self.video_paths[k] for k in indices]\n        batch_labels = [self.labels[k] for k in indices]\n        \n        # Pre-process the videos and store in X, y\n        X = np.empty((self.batch_size, MAX_SEQ_LENGTH, IMG_SIZE, IMG_SIZE, 3))\n        y = np.empty((self.batch_size), dtype=int)\n\n        for i, (path, label) in enumerate(zip(batch_paths, batch_labels)):\n            # Load and pre-process the video\n            frames = extract_faces_from_video(path)\n            X[i,] = frames\n            y[i] = label\n            \n        return X, y\n\n    def on_epoch_end(self):\n        \"\"\"Shuffle indices after each epoch if shuffle is enabled.\"\"\"\n        self.indices = np.arange(len(self.video_paths))\n        if self.shuffle == True:\n            np.random.shuffle(self.indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:20.983103Z","iopub.execute_input":"2025-08-26T19:47:20.983331Z","iopub.status.idle":"2025-08-26T19:47:20.993725Z","shell.execute_reply.started":"2025-08-26T19:47:20.983278Z","shell.execute_reply":"2025-08-26T19:47:20.993099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# 5. Main Execution Block for Training and Prediction\n# ==============================================================================\n\nif __name__ == \"__main__\":\n    # --- Step 5.1: Load Data Paths and Labels ---\n    print(\"Starting data loading...\")\n    video_paths, labels = load_dataset_from_dir(DATA_DIR)\n    print(f\"Loaded {len(video_paths)} video paths and labels.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:20.994750Z","iopub.execute_input":"2025-08-26T19:47:20.994928Z","iopub.status.idle":"2025-08-26T19:47:21.207055Z","shell.execute_reply.started":"2025-08-26T19:47:20.994896Z","shell.execute_reply":"2025-08-26T19:47:21.206334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    # --- Step 5.2: Split Data for Training and Validation ---\n    X_train_paths, X_val_paths, y_train, y_val = train_test_split(\n        video_paths, labels, test_size=0.2, random_state=42, stratify=labels\n    )\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:21.208180Z","iopub.execute_input":"2025-08-26T19:47:21.208489Z","iopub.status.idle":"2025-08-26T19:47:21.213737Z","shell.execute_reply.started":"2025-08-26T19:47:21.208379Z","shell.execute_reply":"2025-08-26T19:47:21.213073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    # --- Step 5.3: Create Data Generators ---\n    train_generator = DataGenerator(X_train_paths, y_train, batch_size=BATCH_SIZE)\n    val_generator = DataGenerator(X_val_paths, y_val, batch_size=BATCH_SIZE, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:21.215871Z","iopub.execute_input":"2025-08-26T19:47:21.216111Z","iopub.status.idle":"2025-08-26T19:47:21.222432Z","shell.execute_reply.started":"2025-08-26T19:47:21.216066Z","shell.execute_reply":"2025-08-26T19:47:21.221781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"  # --- Step 5.4: Create and Train Model ---\n    model = create_model()\n    \n    print(\"\\nStarting model training...\")\n    history = model.fit(\n        train_generator,\n        epochs=EPOCHS,\n        validation_data=val_generator\n    )\n    \n    print(\"\\nTraining complete.\")\n    \n    # Save the trained model for future use\n    model.save(\"deepfake_detector.h5\")\n    print(\"Model saved to 'deepfake_detector.h5'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:21.223593Z","iopub.execute_input":"2025-08-26T19:47:21.223874Z","iopub.status.idle":"2025-08-26T19:47:32.093619Z","shell.execute_reply.started":"2025-08-26T19:47:21.223829Z","shell.execute_reply":"2025-08-26T19:47:32.092435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # --- Step 5.5: Model Evaluation ---\n    print(\"\\nEvaluating model on the validation set...\")\n    y_pred_probs = model.predict(val_generator)\n    y_pred_classes = (y_pred_probs > 0.5).astype(\"int32\")\n    \n    # Flatten y_val since val_generator returns batches.\n    # Note: This is an approximation. For a perfect evaluation, you'd\n    # collect all predictions and labels in one go.\n    y_val_flat = np.concatenate([y for X, y in val_generator], axis=0)\n    \n    accuracy = accuracy_score(y_val_flat, y_pred_classes)\n    precision = precision_score(y_val_flat, y_pred_classes)\n    recall = recall_score(y_val_flat, y_pred_classes)\n    f1 = f1_score(y_val_flat, y_pred_classes)\n    auc = roc_auc_score(y_val_flat, y_pred_probs)\n    \n    print(\"\\nModel Performance Metrics:\")\n    print(f\"Accuracy:  {accuracy:.4f}\")\n    print(f\"Precision: {precision:.4f}\")\n    print(f\"Recall:    {recall:.4f}\")\n    print(f\"F1-Score:  {f1:.4f}\")\n    print(f\"AUC-ROC:   {auc:.4f}\")\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:32.095195Z","iopub.status.idle":"2025-08-26T19:47:32.095806Z","shell.execute_reply":"2025-08-26T19:47:32.095449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # --- Step 5.6: Prediction on a single video ---\n    # This shows how to use the trained model on a new video file.\n    new_video_path = \"/kaggle/input/deepfake-detection-challenge/test_videos/aaxjpsnvrq.mp4\"\n    if os.path.exists(new_video_path):\n        print(f\"\\nMaking a prediction on a new video: {new_video_path}\")\n        new_video_frames = extract_faces_from_video(new_video_path)\n        # Add the batch dimension for prediction\n        new_video_frames = np.expand_dims(new_video_frames, axis=0)\n        \n        # Get the deepfake probability from the model\n        prediction = model.predict(new_video_frames)[0][0]\n        \n        print(\"\\nPrediction:\")\n        if prediction > 0.5:\n            print(f\"This video is likely a DEEPFAKE with confidence {prediction:.2f}\")\n        else:\n            print(f\"This video is likely REAL with confidence {1 - prediction:.2f}\")\n    else:\n        print(f\"\\nPrediction skipped: Could not find the new video at {new_video_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-26T19:47:32.097148Z","iopub.status.idle":"2025-08-26T19:47:32.097745Z","shell.execute_reply":"2025-08-26T19:47:32.097378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}