{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-07T11:47:02.755494Z","iopub.execute_input":"2025-01-07T11:47:02.755861Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.model_selection import KFold\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix, classification_report, precision_recall_fscore_support\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\n\n\n\n\n","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-01-07T12:16:26.609727Z","iopub.execute_input":"2025-01-07T12:16:26.610044Z","iopub.status.idle":"2025-01-07T12:16:33.099983Z","shell.execute_reply.started":"2025-01-07T12:16:26.609997Z","shell.execute_reply":"2025-01-07T12:16:33.098755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dataset\n# Set up the paths for Kaggle dataset\ntrain_dir = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ntest_dir = '/kaggle/input/deepfake-detection-challenge/test_videos'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T12:16:46.357016Z","iopub.execute_input":"2025-01-07T12:16:46.357391Z","iopub.status.idle":"2025-01-07T12:16:46.362218Z","shell.execute_reply.started":"2025-01-07T12:16:46.357325Z","shell.execute_reply":"2025-01-07T12:16:46.361168Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_frames(video_path, frame_rate=30):\n    print(f\"Extracting frames from: {video_path}\")\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    while(cap.isOpened()):\n        ret, frame = cap.read()\n        if not ret:\n            break\n        # Capture every `frame_rate`-th frame\n        if int(cap.get(1)) % frame_rate == 0:\n            frames.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))  # Convert to RGB\n    cap.release()\n    print(f\"Extracted {len(frames)} frames from {video_path}\")\n    return np.array(frames)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T12:16:48.845578Z","iopub.execute_input":"2025-01-07T12:16:48.845909Z","iopub.status.idle":"2025-01-07T12:16:48.856523Z","shell.execute_reply.started":"2025-01-07T12:16:48.845860Z","shell.execute_reply":"2025-01-07T12:16:48.853812Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess the frames: resize and normalize\ndef preprocess_frames(frames, target_size=(128, 128)):\n    print(f\"Preprocessing {len(frames)} frames: resizing and normalizing\")\n    frames_resized = [cv2.resize(frame, target_size) for frame in frames]\n    frames_normalized = np.array(frames_resized) / 255.0  # Normalize pixel values\n    print(f\"Preprocessing completed. Shape of processed frames: {frames_normalized.shape}\")\n    return frames_normalized\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T12:16:51.610758Z","iopub.execute_input":"2025-01-07T12:16:51.611092Z","iopub.status.idle":"2025-01-07T12:16:51.617225Z","shell.execute_reply.started":"2025-01-07T12:16:51.611042Z","shell.execute_reply":"2025-01-07T12:16:51.616040Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and preprocess data\ndef load_data_from_directory(directory_path, frame_rate=30, target_size=(128, 128)):\n    print(f\"Loading data from directory: {directory_path}\")\n    file_paths = [os.path.join(directory_path, f) for f in os.listdir(directory_path) if f.endswith('.mp4')]\n    frames = []\n    labels = []\n    \n    for file_path in file_paths:\n        video_frames = extract_frames(file_path, frame_rate)\n        video_frames = preprocess_frames(video_frames, target_size)\n        frames.extend(video_frames)\n        \n        # Labels: 0 for real, 1 for fake (adjust based on directory structure or metadata)\n        label = 0 if 'real' in file_path else 1\n        labels.extend([label] * len(video_frames))\n    \n    print(f\"Loaded {len(frames)} frames from {len(file_paths)} videos.\")\n    return np.array(frames), np.array(labels)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T12:16:54.160864Z","iopub.execute_input":"2025-01-07T12:16:54.161195Z","iopub.status.idle":"2025-01-07T12:16:54.169671Z","shell.execute_reply.started":"2025-01-07T12:16:54.161147Z","shell.execute_reply":"2025-01-07T12:16:54.168685Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model(input_shape=(128, 128, 3)):\n    model = Sequential()\n    \n    # First Conv Layer with L2 Regularization and Batch Normalization\n    model.add(Conv2D(32, (3, 3), activation='relu', input_shape=input_shape, kernel_regularizer=l2(0.01), name=\"conv1\"))\n    model.add(BatchNormalization())  # Batch Normalization\n    model.add(MaxPooling2D((2, 2), name=\"maxpool1\"))\n    \n    # Second Conv Layer with L2 Regularization and Dropout\n    model.add(Conv2D(64, (3, 3), activation='relu', kernel_regularizer=l2(0.01), name=\"conv2\"))\n    model.add(Dropout(0.5))  # Dropout for regularization\n    model.add(MaxPooling2D((2, 2), name=\"maxpool2\"))\n    \n    model.add(Flatten(name=\"flatten\"))\n    \n    # Dense Layer with Dropout\n    model.add(Dense(128, activation='relu', name=\"dense1\"))\n    model.add(Dropout(0.5))  # Dropout for regularization\n    \n    model.add(Dense(1, activation='sigmoid', name=\"output\"))  # Binary classification: real or fake\n    \n    # Compile the model\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T12:16:56.094188Z","iopub.execute_input":"2025-01-07T12:16:56.094550Z","iopub.status.idle":"2025-01-07T12:16:56.105567Z","shell.execute_reply.started":"2025-01-07T12:16:56.094489Z","shell.execute_reply":"2025-01-07T12:16:56.104096Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\n# Example usage\ndef main():\n    # Load training data\n    X_train, y_train = load_data_from_directory(train_dir, frame_rate=30, target_size=(128, 128))\n    \n    # Split data into training and validation sets\n    X_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n    # Data Augmentation\n    train_datagen = ImageDataGenerator(\n        rescale=1./255,\n        rotation_range=20,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest'\n    )\n\n    # Apply augmentation to the training data\n    train_generator = train_datagen.flow(X_train, y_train, batch_size=32)\n\n    # Create the CNN model\n    model = create_model()\n\n    # Early stopping callback\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n\n    # Train the model with augmented data and store the history\n    history = model.fit(train_generator, epochs=10, validation_data=(X_val, y_val), callbacks=[early_stopping])\n\n    # Evaluate the model on the validation set\n    val_loss, val_acc = model.evaluate(X_val, y_val)\n    print(f\"Validation accuracy: {val_acc}\")\n\n    # Plot training and validation loss and accuracy\n    import matplotlib.pylab as plt\n\n    # Plot training and validation loss\n    plt.figure(figsize=(12, 6))\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    # Plot training and validation accuracy\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['accuracy'], label='Training Accuracy')\n    plt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    # Show the plots\n    plt.tight_layout()\n    plt.show()\n\n# Run the main function\nif __name__ == \"__main__\":\n    main()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T12:21:08.026132Z","iopub.execute_input":"2025-01-07T12:21:08.026518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns\n\n# Set up the paths for Kaggle dataset\ntrain_dir = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ntest_dir = '/kaggle/input/deepfake-detection-challenge/test_videos'\n\n# Function to extract frames\ndef extract_frames(video_path, frame_rate=30):\n    print(f\"Extracting frames from: {video_path}\")\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n        if int(cap.get(1)) % frame_rate == 0:\n            frames.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))\n    cap.release()\n    print(f\"Extracted {len(frames)} frames from {video_path}\")\n    return np.array(frames)\n\n# Preprocess the frames: resize and normalize\ndef preprocess_frames(frames, target_size=(128, 128)):\n    print(f\"Preprocessing {len(frames)} frames: resizing and normalizing\")\n    frames_resized = [cv2.resize(frame, target_size) for frame in frames]\n    frames_normalized = np.array(frames_resized, dtype=np.float32) / 255.0\n    print(f\"Preprocessing completed. Shape of processed frames: {frames_normalized.shape}\")\n    return frames_normalized\n\n# Load and preprocess data\ndef load_data_from_directory(directory_path, frame_rate=30, target_size=(128, 128)):\n    print(f\"Loading data from directory: {directory_path}\")\n    file_paths = [os.path.join(directory_path, f) for f in os.listdir(directory_path) if f.endswith('.mp4')]\n    frames = []\n    labels = []\n    \n    for file_path in file_paths:\n        video_frames = extract_frames(file_path, frame_rate)\n        video_frames = preprocess_frames(video_frames, target_size)\n        frames.extend(video_frames)\n        \n        label = 0 if 'real' in file_path else 1\n        labels.extend([label] * len(video_frames))\n    \n    print(f\"Loaded {len(frames)} frames from {len(file_paths)} videos.\")\n    return np.array(frames), np.array(labels)\n\n# Model creation\ndef create_model(input_shape=(128, 128, 3)):\n    model = Sequential()\n    model.add(Conv2D(32, (3, 3), kernel_regularizer=l2(0.01), name=\"conv1\", input_shape=input_shape))\n    model.add(BatchNormalization())\n    model.add(tf.keras.layers.ReLU())\n    model.add(MaxPooling2D((2, 2), name=\"maxpool1\"))\n    model.add(Conv2D(64, (3, 3), kernel_regularizer=l2(0.01), name=\"conv2\"))\n    model.add(Dropout(0.5))\n    model.add(MaxPooling2D((2, 2), name=\"maxpool2\"))\n    model.add(Flatten(name=\"flatten\"))\n    model.add(Dense(128, activation='relu', name=\"dense1\"))\n    model.add(Dropout(0.5))\n    model.add(Dense(1, activation='sigmoid', name=\"output\"))\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\n# Predict real or fake for a single video\ndef predict_video(video_path, model, frame_rate=30, target_size=(128, 128)):\n    print(f\"Processing video: {video_path}\")\n    frames = extract_frames(video_path, frame_rate=frame_rate)\n    if len(frames) == 0:\n        print(\"No frames extracted. Skipping this video.\")\n        return \"Error: No frames\"\n    preprocessed_frames = preprocess_frames(frames, target_size=target_size)\n    predictions = model.predict(preprocessed_frames)\n    mean_prediction = np.mean(predictions.flatten())\n    print(f\"Mean prediction score: {mean_prediction}\")\n    video_label = \"Fake\" if mean_prediction >= 0.5 else \"Real\"\n    return video_label\n\n# Main function\ndef main():\n    # Load and preprocess training data\n    X_train, y_train = load_data_from_directory(train_dir, frame_rate=30, target_size=(128, 128))\n    X_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n    # Data Augmentation\n    train_datagen = ImageDataGenerator(\n        rotation_range=20,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest'\n    )\n    train_generator = train_datagen.flow(X_train, y_train, batch_size=32)\n    val_datagen = ImageDataGenerator()\n    val_generator = val_datagen.flow(X_val, y_val, batch_size=32)\n\n    # Create model and train\n    model = create_model()\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n    history = model.fit(train_generator, epochs=10, validation_data=val_generator, callbacks=[early_stopping])\n\n    # Evaluate model\n    val_loss, val_acc = model.evaluate(val_generator)\n    print(f\"Validation accuracy: {val_acc}\")\n\n    # Confusion matrix and classification report\n    y_pred = (model.predict(X_val) > 0.5).astype(\"int32\")\n    print(\"Classification Report:\\n\", classification_report(y_val, y_pred))\n    conf_matrix = confusion_matrix(y_val, y_pred)\n    sns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\n    plt.title(\"Confusion Matrix\")\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"Actual\")\n    plt.show()\n\n    # Test on a single video\n    test_video_path = os.path.join(test_dir, '/kaggle/input/deepfake-detection-challenge/test_videos/aassnaulhq.mp4')  # Change this to your test video path\n    video_label = predict_video(test_video_path, model)\n    print(f\"The video '{os.path.basename(test_video_path)}' is predicted to be: {video_label}\")\n\n    # Batch prediction on all test videos\n    video_files = [os.path.join(test_dir, f) for f in os.listdir(test_dir) if f.endswith('.mp4')]\n    for video_path in video_files:\n        video_label = predict_video(video_path, model)\n        print(f\"Video: {os.path.basename(video_path)} - Predicted as: {video_label}\")\n\n# Run the main function\nif __name__ == \"__main__\":\n    main()\n","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-01-07T10:03:28.133589Z","iopub.execute_input":"2025-01-07T10:03:28.133955Z","iopub.status.idle":"2025-01-07T10:31:10.739411Z","shell.execute_reply.started":"2025-01-07T10:03:28.133906Z","shell.execute_reply":"2025-01-07T10:31:10.737917Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**THE NEW SUCESS VERSION WITH HIGH ACCURACY**","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns\n\n# Set up the paths for Kaggle dataset\ntrain_dir = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ntest_dir = '/kaggle/input/deepfake-detection-challenge/test_videos'\n\n# Function to extract frames\ndef extract_frames(video_path, frame_rate=30):\n    print(f\"Extracting frames from: {video_path}\")\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n        if int(cap.get(1)) % frame_rate == 0:\n            frames.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))\n    cap.release()\n    print(f\"Extracted {len(frames)} frames from {video_path}\")\n    return np.array(frames)\n\n# Preprocess the frames: resize and normalize\ndef preprocess_frames(frames, target_size=(128, 128)):\n    print(f\"Preprocessing {len(frames)} frames: resizing and normalizing\")\n    frames_resized = [cv2.resize(frame, target_size) for frame in frames]\n    frames_normalized = np.array(frames_resized, dtype=np.float32) / 255.0\n    print(f\"Preprocessing completed. Shape of processed frames: {frames_normalized.shape}\")\n    return frames_normalized\n\n# Load and preprocess data\ndef load_data_from_directory(directory_path, frame_rate=30, target_size=(128, 128)):\n    print(f\"Loading data from directory: {directory_path}\")\n    file_paths = [os.path.join(directory_path, f) for f in os.listdir(directory_path) if f.endswith('.mp4')]\n    frames = []\n    labels = []\n    \n    for file_path in file_paths:\n        video_frames = extract_frames(file_path, frame_rate)\n        video_frames = preprocess_frames(video_frames, target_size)\n        frames.extend(video_frames)\n        \n        label = 0 if 'real' in file_path else 1\n        labels.extend([label] * len(video_frames))\n    \n    print(f\"Loaded {len(frames)} frames from {len(file_paths)} videos.\")\n    return np.array(frames), np.array(labels)\n\n# Model creation\ndef create_model(input_shape=(128, 128, 3)):\n    model = Sequential()\n    model.add(Conv2D(32, (3, 3), kernel_regularizer=l2(0.01), name=\"conv1\", input_shape=input_shape))\n    model.add(BatchNormalization())\n    model.add(tf.keras.layers.ReLU())\n    model.add(MaxPooling2D((2, 2), name=\"maxpool1\"))\n    model.add(Conv2D(64, (3, 3), kernel_regularizer=l2(0.01), name=\"conv2\"))\n    model.add(Dropout(0.5))\n    model.add(MaxPooling2D((2, 2), name=\"maxpool2\"))\n    model.add(Flatten(name=\"flatten\"))\n    model.add(Dense(128, activation='relu', name=\"dense1\"))\n    model.add(Dropout(0.5))\n    model.add(Dense(1, activation='sigmoid', name=\"output\"))\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\n# Predict real or fake for a single video\ndef predict_video(video_path, model, frame_rate=30, target_size=(128, 128)):\n    print(f\"Processing video: {video_path}\")\n    frames = extract_frames(video_path, frame_rate=frame_rate)\n    if len(frames) == 0:\n        print(\"No frames extracted. Skipping this video.\")\n        return \"Error: No frames\"\n    preprocessed_frames = preprocess_frames(frames, target_size=target_size)\n    predictions = model.predict(preprocessed_frames)\n    mean_prediction = np.mean(predictions.flatten())\n    print(f\"Mean prediction score: {mean_prediction}\")\n    video_label = \"Fake\" if mean_prediction >= 0.5 else \"Real\"\n    return video_label\n\n# Main function\ndef main():\n    # Load and preprocess training data\n    X_train, y_train = load_data_from_directory(train_dir, frame_rate=30, target_size=(128, 128))\n    X_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n    # Data Augmentation\n    train_datagen = ImageDataGenerator(\n        rotation_range=20,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest'\n    )\n    train_generator = train_datagen.flow(X_train, y_train, batch_size=32)\n    val_datagen = ImageDataGenerator()\n    val_generator = val_datagen.flow(X_val, y_val, batch_size=32)\n\n    # Create model and train\n    model = create_model()\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n    history = model.fit(train_generator, epochs=10, validation_data=val_generator, callbacks=[early_stopping])\n\n    # Evaluate model\n    val_loss, val_acc = model.evaluate(val_generator)\n    print(f\"Validation accuracy: {val_acc}\")\n\n    # Plot Training & Validation Accuracy and Loss\n    plt.plot(history.history['accuracy'], label='Train Accuracy')\n    plt.plot(history.history['val_accuracy'], label='Val Accuracy')\n    plt.title('Training vs Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    plt.show()\n\n    plt.plot(history.history['loss'], label='Train Loss')\n    plt.plot(history.history['val_loss'], label='Val Loss')\n    plt.title('Training vs Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.show()\n\n    # Confusion matrix and classification report\n    y_pred = (model.predict(X_val) > 0.5).astype(\"int32\")\n    print(\"Classification Report:\\n\", classification_report(y_val, y_pred))\n    conf_matrix = confusion_matrix(y_val, y_pred)\n    sns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\n    plt.title(\"Confusion Matrix\")\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"Actual\")\n    plt.show()\n\n    # Test on a single video\n    test_video_path = os.path.join(test_dir, '/kaggle/input/deepfake-detection-challenge/test_videos/aassnaulhq.mp4')  # Change this to your test video path\n    video_label = predict_video(test_video_path, model)\n    print(f\"The video '{os.path.basename(test_video_path)}' is predicted to be: {video_label}\")\n\n    # Batch prediction on all test videos\n    video_files = [os.path.join(test_dir, f) for f in os.listdir(test_dir) if f.endswith('.mp4')]\n    for video_path in video_files:\n        video_label = predict_video(video_path, model)\n        print(f\"Video: {os.path.basename(video_path)} - Predicted as: {video_label}\")\n\n# Run the main function\nif __name__ == \"__main__\":\n    main()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pylab as plt\nimport cv2\nplt.style.use('ggplot')\nfrom IPython.display import Video\nfrom IPython.display import HTML\n!ls -GFlash ../input/deepfake-detection-challenge\n!du -sh ../input/deepfake-detection-challenge/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T08:13:57.578421Z","iopub.execute_input":"2025-01-07T08:13:57.578825Z","iopub.status.idle":"2025-01-07T08:13:59.926183Z","shell.execute_reply.started":"2025-01-07T08:13:57.578770Z","shell.execute_reply":"2025-01-07T08:13:59.924851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T08:14:02.933246Z","iopub.execute_input":"2025-01-07T08:14:02.933695Z","iopub.status.idle":"2025-01-07T08:14:03.340627Z","shell.execute_reply.started":"2025-01-07T08:14:02.933622Z","shell.execute_reply":"2025-01-07T08:14:03.339641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\n# Load metadata\nmetadata_path = '../input/deepfake-detection-challenge/train_sample_videos/metadata.json'\ntrain_sample_metadata = pd.read_json(metadata_path).T\n\n# Display the first few rows of the metadata\nprint(\"Metadata Sample:\")\nprint(train_sample_metadata.head())\n\n# Visualize the distribution of fake vs. real videos\nplt.figure(figsize=(8, 6))\nsns.countplot(data=train_sample_metadata, x='label', palette='Set1')\nplt.title('Distribution of Fake vs. Real Videos')\nplt.xlabel('Label (Real vs Fake)')\nplt.ylabel('Number of Videos')\nplt.show()\n\n# Further analysis: Count of videos by category (real/fake)\ncategory_counts = train_sample_metadata['label'].value_counts()\nprint(\"Category Counts:\")\nprint(category_counts)\n\n# Pie chart visualization of class distribution\nplt.figure(figsize=(8, 6))\ncategory_counts.plot.pie(autopct='%1.1f%%', startangle=90, colors=['lightgreen', 'salmon'], legend=False)\nplt.title('Class Distribution (Real vs Fake)')\nplt.ylabel('')\nplt.show()\n\n# Check if the 'duration' column exists\nif 'duration' in train_sample_metadata.columns:\n    # Further analysis: Length of videos based on label (Real vs Fake)\n    train_sample_metadata['duration'] = train_sample_metadata['duration'].astype(float)  # Ensure duration is float\n    real_videos_duration = train_sample_metadata[train_sample_metadata['label'] == 'REAL']['duration']\n    fake_videos_duration = train_sample_metadata[train_sample_metadata['label'] == 'FAKE']['duration']\n\n    plt.figure(figsize=(10, 6))\n    sns.boxplot(x='label', y='duration', data=train_sample_metadata, palette='Set2')\n    plt.title('Video Duration by Label (Real vs Fake)')\n    plt.xlabel('Label')\n    plt.ylabel('Video Duration (seconds)')\n    plt.show()\n\n    # Additional Graph: Duration of Real vs Fake Videos Distribution\n    plt.figure(figsize=(10, 6))\n    sns.histplot(real_videos_duration, color='lightgreen', label='Real Videos', kde=True, bins=30)\n    sns.histplot(fake_videos_duration, color='salmon', label='Fake Videos', kde=True, bins=30)\n    plt.title('Distribution of Video Durations (Real vs Fake)')\n    plt.xlabel('Video Duration (seconds)')\n    plt.ylabel('Frequency')\n    plt.legend()\n    plt.show()\nelse:\n    print(\"The 'duration' column is not present in the metadata. Skipping video duration analysis.\")\n\n# Define a function to display a frame from a video\ndef display_video_frame(video_path):\n    cap = cv2.VideoCapture(video_path)\n    ret, frame = cap.read()\n    cap.release()\n    \n    if ret:\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        plt.imshow(frame_rgb)\n        plt.axis('off')\n        plt.show()\n    else:\n        print(f\"Failed to read video: {video_path}\")\n\n# Path to the video folder\nvideo_folder = '../input/deepfake-detection-challenge/train_sample_videos/'\n\n# Display some sample frames from fake videos\nfake_videos = train_sample_metadata[train_sample_metadata['label'] == 'FAKE'].index\nprint(\"Sample frames from Fake videos:\")\nfor video in fake_videos[:3]:  # Display the first 3 fake videos\n    print(f\"Video: {video}\")\n    display_video_frame(os.path.join(video_folder, video))\n\n# Display some sample frames from real videos\nreal_videos = train_sample_metadata[train_sample_metadata['label'] == 'REAL'].index\nprint(\"Sample frames from Real videos:\")\nfor video in real_videos[:3]:  # Display the first 3 real videos\n    print(f\"Video: {video}\")\n    display_video_frame(os.path.join(video_folder, video))\n\n# Additional Analysis: Real vs Fake Videos in Terms of Aspect Ratio\ndef get_video_aspect_ratio(video_path):\n    cap = cv2.VideoCapture(video_path)\n    ret, frame = cap.read()\n    cap.release()\n    \n    if ret:\n        height, width, _ = frame.shape\n        return width / height\n    return None\n\n# Get aspect ratio for real and fake videos\nreal_videos_aspect_ratio = [get_video_aspect_ratio(os.path.join(video_folder, video)) for video in real_videos]\nfake_videos_aspect_ratio = [get_video_aspect_ratio(os.path.join(video_folder, video)) for video in fake_videos]\n\n# Remove None values (in case of failed aspect ratio extraction)\nreal_videos_aspect_ratio = [ratio for ratio in real_videos_aspect_ratio if ratio is not None]\nfake_videos_aspect_ratio = [ratio for ratio in fake_videos_aspect_ratio if ratio is not None]\n\n# Plot aspect ratios of real vs fake videos\nplt.figure(figsize=(10, 6))\nsns.kdeplot(real_videos_aspect_ratio, color='lightgreen', label='Real Videos', shade=True)\nsns.kdeplot(fake_videos_aspect_ratio, color='salmon', label='Fake Videos', shade=True)\nplt.title('Aspect Ratio Distribution of Real vs Fake Videos')\nplt.xlabel('Aspect Ratio (Width/Height)')\nplt.ylabel('Density')\nplt.legend()\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T08:14:06.507091Z","iopub.execute_input":"2025-01-07T08:14:06.507418Z","iopub.status.idle":"2025-01-07T08:14:33.706532Z","shell.execute_reply.started":"2025-01-07T08:14:06.507372Z","shell.execute_reply":"2025-01-07T08:14:33.705326Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install mtcnn tensorflow opencv-python pandas numpy scikit-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T08:14:37.869799Z","iopub.execute_input":"2025-01-07T08:14:37.870166Z","iopub.status.idle":"2025-01-07T08:14:43.717799Z","shell.execute_reply.started":"2025-01-07T08:14:37.870098Z","shell.execute_reply":"2025-01-07T08:14:43.716884Z"},"collapsed":true,"jupyter":{"outputs_hidden":true,"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom mtcnn import MTCNN\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score, classification_report\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n\n# Initialize MTCNN face detector\ndetector = MTCNN()\n\n# Paths and setup\nvideo_folder = '../input/deepfake-detection-challenge/train_sample_videos/'\nmetadata_path = '../input/deepfake-detection-challenge/train_sample_videos/metadata.json'\n\n# Load metadata\ntrain_sample_metadata = pd.read_json(metadata_path).T\n\n# Split metadata into training and validation sets\ntrain_metadata, val_metadata = train_test_split(train_sample_metadata, test_size=0.2, random_state=42)\n\nclass VideoFrameGenerator(Sequence):\n    def __init__(self, metadata, batch_size=32, target_size=(224, 224), shuffle=True):\n        self.metadata = metadata\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.metadata))\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.metadata) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        batch_metadata = self.metadata.iloc[batch_indexes]\n        \n        X, y_labels = self.__data_generation(batch_metadata)\n        return X, y_labels\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n    \n    def __data_generation(self, batch_metadata):\n        X = []\n        y_labels = []\n        \n        for video_name, row in batch_metadata.iterrows():\n            video_path = os.path.join(video_folder, video_name)\n            label = 1 if row['label'] == 'FAKE' else 0\n            \n            cap = cv2.VideoCapture(video_path)\n            while cap.isOpened():\n                ret, frame = cap.read()\n                if not ret:\n                    break\n                \n                frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n                faces = detector.detect_faces(frame_rgb)\n                \n                for face in faces:\n                    x, y, width, height = face['box']\n                    face_img = frame_rgb[y:y+height, x:x+width]\n                    face_img = cv2.resize(face_img, self.target_size)  # Resize face to target_size\n                    face_array = img_to_array(face_img) / 255.0  # Normalize pixel values\n                    \n                    X.append(face_array)\n                    y_labels.append(label)\n                    \n                    if len(X) >= self.batch_size:\n                        cap.release()\n                        return np.array(X), np.array(y_labels)\n            \n            cap.release()\n        \n        # If we exit the loop and don't have enough samples, pad with the first few samples\n        while len(X) < self.batch_size:\n            X.append(X[0])\n            y_labels.append(y_labels[0])\n        \n        return np.array(X), np.array(y_labels)\n\n# Instantiate the generators\nbatch_size = 32\ntrain_generator = VideoFrameGenerator(train_metadata, batch_size=batch_size)\nval_generator = VideoFrameGenerator(val_metadata, batch_size=batch_size)\n\n# Build the model (with deeper CNN layers)\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(256, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(512, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(512, activation='relu'),\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(\n    train_generator,\n    epochs=10,\n    validation_data=val_generator\n)\n\n# Evaluate the model \nloss, accuracy = model.evaluate(val_generator)\nprint(f\"Validation Loss: {loss}\")\nprint(f\"Validation Accuracy: {accuracy}\")\n\n# Perform predictions on the validation set\ny_true = []\ny_pred = []\n\nfor X_batch, y_batch in val_generator:\n    y_true.extend(y_batch)\n    y_pred_batch = model.predict(X_batch)\n    y_pred.extend((y_pred_batch > 0.5).astype(int))  # Convert predictions to binary (0 or 1)\n\n# Classification report\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_true, y_pred))\n\n# You can also calculate precision, recall, and F1-score individually:\nprecision = precision_score(y_true, y_pred)\nrecall = recall_score(y_true, y_pred)\nf1 = f1_score(y_true, y_pred)\n\nprint(f\"\\nPrecision: {precision}\")\nprint(f\"Recall: {recall}\")\nprint(f\"F1-score: {f1}\")\n\n# Plot the training and validation accuracy and loss\nplt.figure(figsize=(12, 5))\n\n# Accuracy plot\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Accuracy over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# Loss plot\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Loss over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n\n   \n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T11:11:40.580454Z","iopub.execute_input":"2025-01-07T11:11:40.581097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    # Plot training and validation loss and accuracy\n    import matplotlib.pylab as plt\n\n    # Plot training and validation loss\n    plt.figure(figsize=(12, 6))\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    # Plot training and validation accuracy\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['accuracy'], label='Training Accuracy')\n    plt.plot(history.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T08:42:08.264922Z","iopub.execute_input":"2025-01-07T08:42:08.265268Z","iopub.status.idle":"2025-01-07T08:42:09.013040Z","shell.execute_reply.started":"2025-01-07T08:42:08.265214Z","shell.execute_reply":"2025-01-07T08:42:09.011463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom mtcnn import MTCNN\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import img_to_array\n\n# Initialize MTCNN face detector\ndetector = MTCNN()\n\n# Load the trained model\n# model = load_model('path/to/your/trained_model.h5')  # Update with the actual path to your model\n\n# Function to detect and preprocess faces from a video\ndef extract_faces_from_video(video_path, target_size=(224, 224)):\n    cap = cv2.VideoCapture(video_path)\n    faces = []\n\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        detected_faces = detector.detect_faces(frame_rgb)\n\n        for face in detected_faces:\n            x, y, width, height = face['box']\n            face_img = frame_rgb[y:y+height, x:x+width]\n            face_img = cv2.resize(face_img, target_size)\n            face_array = img_to_array(face_img) / 255.0\n            faces.append(face_array)\n    \n    cap.release()\n    return np.array(faces)\n\n# Function to predict if the video is fake or real\ndef predict_video(video_path):\n    faces = extract_faces_from_video(video_path)\n    if len(faces) == 0:\n        print(\"No faces detected in the video.\")\n        return None\n\n    predictions = model.predict(faces)\n    avg_prediction = np.mean(predictions)\n\n    if avg_prediction > 0.5:\n        print(f\"The video '{video_path}' is predicted to be FAKE.\")\n    else:\n        print(f\"The video '{video_path}' is predicted to be REAL.\")\n\n    return avg_prediction\n\n# Test the prediction function with a sample video\nvideo_path = '/kaggle/input/deepfake-detection-challenge/test_videos'  # Update with the actual path to the test video\npredict_video(video_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}