{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8751657,"sourceType":"datasetVersion","datasetId":5256654}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport tensorflow as tf\n\n# Function to load all frames from a video directory and associate labels\ndef load_frames_with_labels(video_dir, label, num_classes=2):\n    frames = []\n    for frame_file in sorted(os.listdir(video_dir)):\n        frame_path = os.path.join(video_dir, frame_file)\n        if frame_file.endswith('.jpg') or frame_file.endswith('.png'):  # Adjust as per your frame format\n            image = tf.io.read_file(frame_path)\n            image = tf.image.decode_image(image, channels=3)\n            image = tf.image.resize(image, [128, 128])  # Resize to desired dimensions\n            #image = image / 255.\n            \n            label_one_hot = tf.one_hot(label, num_classes)  # One-hot encode the label\n            frames.append((image, label_one_hot))  # Append frame with its one-hot encoded label\n    return frames\n\n# Function to create dataset with labels for each frame\ndef create_dataset_with_labels(fake_dir, real_dir, max_videos=200, split_ratio=0.8):\n    labeled_frames = []\n    train_frames = []\n    test_frames = []\n\n    # Load fake videos\n    fake_count = 0\n    for video_name in os.listdir(fake_dir):\n        if fake_count >= max_videos:\n            break  # Stop if we have enough fake videos\n        video_path = os.path.join(fake_dir, video_name)\n        if os.path.isdir(video_path):\n            # Load frames and associate label 1 for fake\n            frames_with_labels = load_frames_with_labels(video_path, 1)\n            split_index = int(len(frames_with_labels) * split_ratio)\n            train_frames.extend(frames_with_labels[:split_index])\n            test_frames.extend(frames_with_labels[split_index:])\n            labeled_frames.extend(frames_with_labels)  # Add to the dataset\n            fake_count += 1  # Increment the count of fake videos\n            print(f'Loaded fake video {fake_count}')\n\n    # Load real videos\n    real_count = 0\n    for video_name in os.listdir(real_dir):\n        if real_count >= max_videos:\n            break  # Stop if we have enough real videos\n        video_path = os.path.join(real_dir, video_name)\n        if os.path.isdir(video_path):\n            # Load frames and associate label 0 for real\n            frames_with_labels = load_frames_with_labels(video_path, 0)\n            split_index = int(len(frames_with_labels) * split_ratio)\n            train_frames.extend(frames_with_labels[:split_index])\n            test_frames.extend(frames_with_labels[split_index:])\n            labeled_frames.extend(frames_with_labels)  # Add to the dataset\n            real_count += 1  # Increment the count of real videos\n            print(f'Loaded real video {real_count}')\n\n    return labeled_frames, train_frames, test_frames\n\n# Define paths\nBASE_DIR = '/kaggle/input/dfdc-facial-cropped-videos-dataset-jpg-frames/DFDC'\nFAKE_DIR = os.path.join(BASE_DIR, 'FAKE', 'TRAIN')\nREAL_DIR = os.path.join(BASE_DIR, 'REAL', 'TRAIN')\n\n# Create dataset with labels for each frame, loading up to 200 video folders from both categories\nlabeled_dataset, train_dataset, test_dataset = create_dataset_with_labels(FAKE_DIR, REAL_DIR, max_videos=200, split_ratio=0.8)","metadata":{"execution":{"iopub.status.busy":"2024-09-03T13:57:28.512615Z","iopub.execute_input":"2024-09-03T13:57:28.513177Z","iopub.status.idle":"2024-09-03T14:00:28.845659Z","shell.execute_reply.started":"2024-09-03T13:57:28.513136Z","shell.execute_reply":"2024-09-03T14:00:28.844501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert to TensorFlow Dataset format\ndef create_tf_dataset(labeled_data):\n    # Unzip the labeled_data into frames and labels\n    frames, labels = zip(*labeled_data)\n    \n    # Create a TensorFlow Dataset from the frames and labels\n    tf_dataset = tf.data.Dataset.from_tensor_slices((tf.stack(frames), tf.convert_to_tensor(labels)))\n    \n    return tf_dataset\n\n# Create TensorFlow datasets\ntf_dataset = create_tf_dataset(labeled_dataset)\ntrain_tf_dataset = create_tf_dataset(train_dataset)\ntest_tf_dataset = create_tf_dataset(test_dataset)\n\n# Shuffle the datasets\ntf_dataset = tf_dataset.shuffle(buffer_size=len(labeled_dataset), reshuffle_each_iteration=True)\ntrain_tf_dataset = train_tf_dataset.shuffle(buffer_size=len(train_dataset), reshuffle_each_iteration=True)\ntest_tf_dataset = test_tf_dataset.shuffle(buffer_size=len(test_dataset), reshuffle_each_iteration=True)\n\n# Batch the datasets\ntf_dataset = tf_dataset.batch(32).prefetch(tf.data.AUTOTUNE)\ntrain_tf_dataset = train_tf_dataset.batch(32).prefetch(tf.data.AUTOTUNE)\ntest_tf_dataset = test_tf_dataset.batch(32).prefetch(tf.data.AUTOTUNE)\n\n# Example of checking the shape of the loaded frames and their labels\nprint(\"Full dataset:\")\nfor frame, label in tf_dataset.take(5):  # Display the first 5 frames and their labels\n    print(f'Frame shape: {frame.shape}, Label: {label.numpy()}')\n\nprint(\"\\nTraining dataset:\")\nfor frame, label in train_tf_dataset.take(5):  # Display the first 5 frames and their labels\n    print(f'Frame shape: {frame.shape}, Label: {label.numpy()}')\n\nprint(\"\\nTesting dataset:\")\nfor frame, label in test_tf_dataset.take(5):  # Display the first 5 frames and their labels\n    print(f'Frame shape: {frame.shape}, Label: {label.numpy()}')","metadata":{"execution":{"iopub.status.busy":"2024-09-03T14:00:28.847708Z","iopub.execute_input":"2024-09-03T14:00:28.848434Z","iopub.status.idle":"2024-09-03T14:00:48.671155Z","shell.execute_reply.started":"2024-09-03T14:00:28.848388Z","shell.execute_reply":"2024-09-03T14:00:48.670154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Function to plot frames\ndef plot_frames(frames, labels):\n    num_frames = 5  # Number of frames in the batch\n    fig, axes = plt.subplots(1, num_frames, figsize=(15, 5))\n    \n    for i in range(num_frames):\n        axes[i].imshow(frames[i].numpy().astype(\"uint8\"))  # Convert tensor to numpy array for plotting\n        axes[i].axis('off')  # Hide axes\n        axes[i].set_title(f'Label: {labels[i].numpy()}')  # Set title with label\n\n    plt.show()\n\n# Example of iterating through the dataset to plot frames\nfor frames, labels in tf_dataset.take(1):  # Take one batch from the dataset\n    print(\"Frames shape:\", frames.shape)  # Should be (batch_size, 128, 128, 3)\n    print(\"Labels:\", labels.numpy())\n    \n    # Plot the frames\n    plot_frames(frames, labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-03T14:01:01.067439Z","iopub.execute_input":"2024-09-03T14:01:01.068155Z","iopub.status.idle":"2024-09-03T14:01:05.527119Z","shell.execute_reply.started":"2024-09-03T14:01:01.068105Z","shell.execute_reply":"2024-09-03T14:01:05.526217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0,EfficientNetB4,EfficientNetB7,EfficientNetB3\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import RandomZoom, RandomBrightness, RandomRotation, RandomFlip, RandomContrast\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndef create_base_learner(model_cls, input_shape=(128, 128, 3), num_classes=2):\n    # Create the base model\n        \n    base_model = model_cls(include_top=False, weights='imagenet', input_shape=input_shape)\n    \n    # Freeze all layers except the new ones\n    \n    base_model.trainable = True\n    \n    for layer in base_model.layers[:-10]:\n        \n        layer.trainable = False\n    \n    # Define the data augmentation layers\n    data_aug = tf.keras.Sequential([\n        RandomFlip('horizontal'),\n        RandomFlip('vertical'),\n        RandomZoom(0.1),\n        RandomRotation(0.1),\n        RandomBrightness(0.1),\n        RandomContrast(0.1)\n    ], name=\"data_aug\")\n    \n   \n    \n    # Apply data augmentation to the input\n    inputs = Input(shape=input_shape)\n    x = data_aug(inputs)\n    x = base_model(x)\n    x = GlobalAveragePooling2D()(x)\n    outputs = Dense(num_classes, activation='softmax')(x)\n    model = Model(inputs=inputs, outputs=outputs)\n    \n    \n        \n    return model\n\n# Initialize the base learners\nbase_learners = [\n    create_base_learner(EfficientNetB0),\n    create_base_learner(EfficientNetB3),\n    create_base_learner(EfficientNetB4),\n    create_base_learner(EfficientNetB7)\n]\n\n# Compile each model with appropriate loss and optimizer\nfor model in base_learners:\n    model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy','precision','auc','recall'])","metadata":{"execution":{"iopub.status.busy":"2024-09-03T14:02:55.972247Z","iopub.execute_input":"2024-09-03T14:02:55.972644Z","iopub.status.idle":"2024-09-03T14:03:35.957530Z","shell.execute_reply.started":"2024-09-03T14:02:55.972607Z","shell.execute_reply":"2024-09-03T14:03:35.956721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit each model\ntf.keras.backend.clear_session()\n\nfor i, model in enumerate(base_learners):\n    print(f\"Training model {i + 1}/{len(base_learners)}: {model.name}\")\n    history = model.fit(\n        train_tf_dataset,    # Your training dataset\n        validation_data=test_tf_dataset,  # Your validation dataset\n        epochs=20,           # Number of epochs\n        batch_size=32,       # Batch size (optional, if using dataset)\n        verbose=1            # Verbosity mode\n    )\n    print(f\"Finished training model {i + 1}/{len(base_learners)}: {model.name}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-03T14:03:45.954210Z","iopub.execute_input":"2024-09-03T14:03:45.954602Z","iopub.status.idle":"2024-09-03T15:06:24.390974Z","shell.execute_reply.started":"2024-09-03T14:03:45.954563Z","shell.execute_reply":"2024-09-03T15:06:24.389983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optionally, you can save each model's weights after training\n\nfor i, model in enumerate(base_learners):\n    model.save_weights(f'/kaggle/working/efficientnet_model_{i + 1}_weights.weights.h5')","metadata":{"execution":{"iopub.status.busy":"2024-09-03T15:09:06.097081Z","iopub.execute_input":"2024-09-03T15:09:06.098006Z","iopub.status.idle":"2024-09-03T15:09:09.686388Z","shell.execute_reply.started":"2024-09-03T15:09:06.097969Z","shell.execute_reply":"2024-09-03T15:09:09.685396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define paths\nBASE_DIR = '/kaggle/input/dfdc-facial-cropped-videos-dataset-jpg-frames/DFDC'\nFAKE_DIR = os.path.join(BASE_DIR, 'FAKE', 'TEST')\nREAL_DIR = os.path.join(BASE_DIR, 'REAL', 'TEST')\n\n# Create dataset with labels for each frame, loading up to 200 video folders from both categories\nlabeled_dataset_2, train_dataset_2, test_dataset_2 = create_dataset_with_labels(FAKE_DIR, REAL_DIR, max_videos=200, split_ratio=0.8)","metadata":{"execution":{"iopub.status.busy":"2024-09-03T15:12:10.535239Z","iopub.execute_input":"2024-09-03T15:12:10.535748Z","iopub.status.idle":"2024-09-03T15:12:41.561400Z","shell.execute_reply.started":"2024-09-03T15:12:10.535706Z","shell.execute_reply":"2024-09-03T15:12:41.560139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create TensorFlow datasets\ntf_dataset_2 = create_tf_dataset(labeled_dataset_2)\ntrain_tf_dataset_2 = create_tf_dataset(train_dataset_2)\ntest_tf_dataset_2 = create_tf_dataset(test_dataset_2)\n\n# Shuffle the datasets\ntf_dataset_2 = tf_dataset_2.shuffle(buffer_size=len(labeled_dataset_2), reshuffle_each_iteration=True)\ntrain_tf_dataset_2 = train_tf_dataset_2.shuffle(buffer_size=len(train_dataset_2), reshuffle_each_iteration=True)\ntest_tf_dataset_2 = test_tf_dataset_2.shuffle(buffer_size=len(test_dataset_2), reshuffle_each_iteration=True)\n\n# Batch the datasets\ntf_dataset_2 = tf_dataset_2.batch(32).prefetch(tf.data.AUTOTUNE)\ntrain_tf_dataset_2 = train_tf_dataset_2.batch(32).prefetch(tf.data.AUTOTUNE)\ntest_tf_dataset_2= test_tf_dataset_2.batch(32).prefetch(tf.data.AUTOTUNE)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Concatenate\n\ndef create_deepfake_stack(base_learners, input_shape=(128, 128, 3), num_classes=2):\n    # Input layer for the base learners\n    inputs = [Input(shape=input_shape) for _ in base_learners]\n    \n    # Get the outputs from each base learner\n    base_outputs = [model(input_tensor) for model, input_tensor in zip(base_learners, inputs)]\n    \n    # Concatenate the outputs of all base learners\n    concatenated = Concatenate()(base_outputs)\n    \n    # Add a Dense layer on top of the concatenated outputs\n    x = Dense(128, activation='relu')(concatenated)\n    x = Dense(64, activation='relu')(x)\n    outputs = Dense(num_classes, activation='softmax')(x)\n    \n    # Create the model\n    model = Model(inputs=inputs, outputs=outputs)\n    \n    return model\n\n# Create the DeepfakeStack model\ndeepfake_stack = create_deepfake_stack(base_learners)\n\n# Compile the DeepfakeStack model\ndeepfake_stack.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy','precision','auc','recall'])\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming `train_tf_dataset` is your training dataset and `test_tf_dataset` is your testing dataset\nhistory = deepfake_stack.fit([train_tf_dataset] * len(base_learners),  # Input dataset replicated for each base learner\n                             epochs=100,  # Adjust the number of epochs as needed\n                             validation_data=([test_tf_dataset] * len(base_learners)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optionally, save the model after training\ndeepfake_stack.save('deepfake_stack_classifier.h5')","metadata":{},"execution_count":null,"outputs":[]}]}