{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":9370003,"sourceType":"datasetVersion","datasetId":5682694}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-10T19:40:34.193141Z","iopub.execute_input":"2024-09-10T19:40:34.193833Z","iopub.status.idle":"2024-09-10T19:40:34.403090Z","shell.execute_reply.started":"2024-09-10T19:40:34.193788Z","shell.execute_reply":"2024-09-10T19:40:34.402171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Data Preparation","metadata":{}},{"cell_type":"code","source":"!pip install keras-nightly","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pickle\nfrom collections import defaultdict\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\n\n# Constants\nIMG_SIZE = (299, 299)  # Target image size for InceptionV3\nMAX_FRAMES = 10        # Maximum number of frames per sequence\nREAL_DIR = 'real'      # Directory containing images of real samples\nFAKE_DIR = 'fake'      # Directory containing images of fake samples\n\ndef load_images_from_directory(directory, label):\n    \"\"\"\n    Load images from the specified directory, group them by videoname, preprocess, and pad sequences.\n    \n    Parameters:\n    - directory (str): Path to the directory containing images.\n    - label (int): Label for the samples (0 for real, 1 for fake).\n    \n    Returns:\n    - data (np.array): Array of processed sequences of images.\n    - labels (np.array): Array of corresponding labels.\n    \"\"\"\n    data = []\n    labels = []\n    video_frames = defaultdict(list)  # Dictionary to hold frames grouped by videoname\n\n    # Iterate over all images in the directory\n    for filename in os.listdir(directory):\n        if filename.endswith('.png'):\n            # Extract the videoname (before the first underscore)\n            video_name = filename.split('_')[0]\n            filepath = os.path.join(directory, filename)\n            img = cv2.imread(filepath)\n\n            if img is not None:\n                # Resize and preprocess the image\n                img = cv2.resize(img, IMG_SIZE)\n                img = img_to_array(img)\n                img = preprocess_input(img)  # Preprocess using InceptionV3 preprocessing\n                video_frames[video_name].append(img)\n    \n    # Process each set of images grouped by videoname\n    for frames in video_frames.values():\n        # Pad with zeros if frames are less than MAX_FRAMES\n        while len(frames) < MAX_FRAMES:\n            frames.append(np.zeros((299, 299, 3)))  # Zero-padding for missing frames\n\n        # Limit to MAX_FRAMES if more frames are present\n        frames = frames[:MAX_FRAMES]\n\n        data.append(frames)\n        labels.append(label)\n    \n    return np.array(data), np.array(labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-12T18:33:13.570285Z","iopub.execute_input":"2024-09-12T18:33:13.570657Z","iopub.status.idle":"2024-09-12T18:33:27.273373Z","shell.execute_reply.started":"2024-09-12T18:33:13.570620Z","shell.execute_reply":"2024-09-12T18:33:27.272392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load real and fake images\nREAL_DIR = \"/kaggle/input/deepfake-detection-challenge-dataset-face-images/real\"\nFAKE_DIR = \"/kaggle/input/deepfake-detection-challenge-dataset-face-images/fake\"\n# Load real and fake images\nx_real, y_real = load_images_from_directory(REAL_DIR, label=0)  # Label 0 for real\nx_fake, y_fake = load_images_from_directory(FAKE_DIR, label=1)  # Label 1 for fake\n","metadata":{"execution":{"iopub.status.busy":"2024-09-12T18:33:30.920479Z","iopub.execute_input":"2024-09-12T18:33:30.921121Z","iopub.status.idle":"2024-09-12T18:33:50.127496Z","shell.execute_reply.started":"2024-09-12T18:33:30.921080Z","shell.execute_reply":"2024-09-12T18:33:50.126471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Combine real and fake data\nx_data = np.concatenate([x_real, x_fake], axis=0)\ny_data = np.concatenate([y_real, y_fake], axis=0)\n\n# Split the data into training and validation sets\nx_train, x_val, y_train, y_val = train_test_split(x_data, y_data, test_size=0.2, random_state=42, stratify=y_data)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-12T18:33:54.272515Z","iopub.execute_input":"2024-09-12T18:33:54.272898Z","iopub.status.idle":"2024-09-12T18:33:55.422831Z","shell.execute_reply.started":"2024-09-12T18:33:54.272859Z","shell.execute_reply":"2024-09-12T18:33:55.422031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import InceptionV3\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, LSTM, GRU, TimeDistributed, Dropout, GlobalAveragePooling2D, Input\n\n# Define the custom feature extractor as a Keras Model\ndef create_feature_extractor():\n    base_model = InceptionV3(weights='imagenet', include_top=False, input_shape=(299, 299, 3))\n    feature_extractor = tf.keras.Model(inputs=base_model.input, outputs=GlobalAveragePooling2D()(base_model.output))\n    return feature_extractor\n\n# Build the sequential model with TimeDistributed\ndef create_model():\n    feature_extractor = create_feature_extractor()\n    \n    model = Sequential()\n    model.add(Input(shape=(None, 299, 299, 3)))\n    model.add(TimeDistributed(feature_extractor))\n    model.add(LSTM(128, return_sequences=True)) # LSTM layer\n    model.add(GRU(128)) # GRU layer\n    model.add(Dropout(0.5)) # Dropout layer for regularization\n    model.add(Dense(64, activation='relu'))\n    model.add(Dense(1, activation='sigmoid'))  # Output layer: binary classification\n    \n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\n# Create and summarize the model\nmodel = create_model()\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-11T20:42:33.647803Z","iopub.execute_input":"2024-09-11T20:42:33.648686Z","iopub.status.idle":"2024-09-11T20:42:37.122971Z","shell.execute_reply.started":"2024-09-11T20:42:33.648645Z","shell.execute_reply":"2024-09-11T20:42:37.122086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train, y_train, epochs=20, batch_size=10, validation_data=(x_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2024-09-11T20:42:52.210696Z","iopub.execute_input":"2024-09-11T20:42:52.211529Z","iopub.status.idle":"2024-09-11T20:56:30.628165Z","shell.execute_reply.started":"2024-09-11T20:42:52.211487Z","shell.execute_reply":"2024-09-11T20:56:30.627210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model in the .h5 format\nmodel.save('/kaggle/working/deepfake_detection_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-09-11T20:56:43.783162Z","iopub.execute_input":"2024-09-11T20:56:43.783551Z","iopub.status.idle":"2024-09-11T20:56:44.835969Z","shell.execute_reply.started":"2024-09-11T20:56:43.783514Z","shell.execute_reply":"2024-09-11T20:56:44.835170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Load the model\nmodel_path = '/kaggle/working/deepfake_detection_model.h5'\nmodel = tf.keras.models.load_model(model_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-11T21:02:38.398338Z","iopub.execute_input":"2024-09-11T21:02:38.398769Z","iopub.status.idle":"2024-09-11T21:02:39.639973Z","shell.execute_reply.started":"2024-09-11T21:02:38.398728Z","shell.execute_reply":"2024-09-11T21:02:39.638547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save model architecture\nmodel_json = model.to_json()\nwith open('/kaggle/working/model_architecture.json', 'w') as json_file:\n    json_file.write(model_json)\n\n# Save model weights\nmodel.save_weights('/kaggle/working/model.weights.h5')","metadata":{"execution":{"iopub.status.busy":"2024-09-11T20:58:58.429720Z","iopub.execute_input":"2024-09-11T20:58:58.430135Z","iopub.status.idle":"2024-09-11T20:58:59.598197Z","shell.execute_reply.started":"2024-09-11T20:58:58.430090Z","shell.execute_reply":"2024-09-11T20:58:59.597437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import model_from_json\n\n# Load model architecture\nwith open('/kaggle/working/model_architecture.json', 'r') as json_file:\n    model_json = json_file.read()\n\n# Recreate the model from the architecture\nmodel = model_from_json(model_json, custom_objects={'TimeDistributed': tf.keras.layers.TimeDistributed})\n\n# Load model weights\nmodel.load_weights('/kaggle/working/model.weights.h5')\nmodel.compile(optimizer='rmsprop', loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-09-11T20:59:14.511612Z","iopub.execute_input":"2024-09-11T20:59:14.512488Z","iopub.status.idle":"2024-09-11T20:59:16.569775Z","shell.execute_reply.started":"2024-09-11T20:59:14.512450Z","shell.execute_reply":"2024-09-11T20:59:16.568705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\n\n# Load the trained model\n#model = load_model('/kaggle/working/deepfake_detection_model_2.h5')\n\n# Constants\nIMG_SIZE = (299, 299)  # Image size expected by the model\nMAX_FRAMES = 10        # Max frames to consider per video\n\ndef extract_frames_from_video(video_path, max_frames=MAX_FRAMES):\n    \"\"\"\n    Extract and preprocess frames from a given video for prediction.\n    \n    Parameters:\n    - video_path (str): Path to the input video file.\n    - max_frames (int): Maximum number of frames to process from the video.\n\n    Returns:\n    - processed_frames (np.array): Array of processed frames ready for model prediction.\n    \"\"\"\n    cap = cv2.VideoCapture(video_path)\n    frame_count = 0\n    processed_frames = []\n\n    while cap.isOpened() and frame_count < max_frames:\n        ret, frame = cap.read()\n        if not ret:\n            break\n        \n        # Resize and preprocess the frame\n        resized_frame = cv2.resize(frame, IMG_SIZE)\n        frame_array = img_to_array(resized_frame)\n        processed_frame = preprocess_input(frame_array)\n        processed_frames.append(processed_frame)\n        frame_count += 1\n\n    cap.release()\n\n    # Pad with zero frames if less than max_frames are present\n    while len(processed_frames) < max_frames:\n        processed_frames.append(np.zeros((299, 299, 3)))\n    \n    return np.array([processed_frames])\n\ndef predict_video(model, video_path):\n    \"\"\"\n    Predict whether the video is REAL or FAKE based on extracted frames.\n\n    Parameters:\n    - model: The trained deepfake detection model.\n    - video_path (str): Path to the input video file.\n\n    Returns:\n    - prediction (str): 'REAL' or 'FAKE' based on model prediction.\n    \"\"\"\n    # Extract frames from the video\n    frames = extract_frames_from_video(video_path)\n    \n    # Make predictions on frames\n    predictions = model.predict(frames)\n    print(predictions)\n    # Aggregate predictions; if the average is above 0.5, classify as FAKE\n    avg_prediction = np.mean(predictions)\n    print(avg_prediction)\n    if avg_prediction > 0.5:\n        return 'FAKE'\n    else:\n        return 'REAL'","metadata":{"execution":{"iopub.status.busy":"2024-09-10T20:04:22.196229Z","iopub.execute_input":"2024-09-10T20:04:22.197108Z","iopub.status.idle":"2024-09-10T20:04:22.207696Z","shell.execute_reply.started":"2024-09-10T20:04:22.197070Z","shell.execute_reply":"2024-09-10T20:04:22.206618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage\nvideo_path = '/kaggle/input/deepfake-detection-challenge/test_videos/bcbqxhziqz.mp4'  # Replace with your video file path\nresult = predict_video(model, video_path)\nprint(f'The video is predicted to be: {result}')","metadata":{"execution":{"iopub.status.busy":"2024-09-10T20:04:23.074312Z","iopub.execute_input":"2024-09-10T20:04:23.074662Z","iopub.status.idle":"2024-09-10T20:04:23.441897Z","shell.execute_reply.started":"2024-09-10T20:04:23.074630Z","shell.execute_reply":"2024-09-10T20:04:23.441046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}