{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-20T11:26:47.162234Z","iopub.execute_input":"2024-05-20T11:26:47.162454Z","iopub.status.idle":"2024-05-20T11:26:47.426162Z","shell.execute_reply.started":"2024-05-20T11:26:47.162416Z","shell.execute_reply":"2024-05-20T11:26:47.425309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pylab as plt\nimport cv2\nplt.style.use('ggplot')\nfrom IPython.display import Video\nfrom IPython.display import HTML","metadata":{"execution":{"iopub.status.busy":"2024-05-20T11:26:48.299900Z","iopub.execute_input":"2024-05-20T11:26:48.300270Z","iopub.status.idle":"2024-05-20T11:26:48.326292Z","shell.execute_reply.started":"2024-05-20T11:26:48.300207Z","shell.execute_reply":"2024-05-20T11:26:48.325681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -GFlash ../input/deepfake-detection-challenge\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T11:26:49.691291Z","iopub.execute_input":"2024-05-20T11:26:49.691640Z","iopub.status.idle":"2024-05-20T11:26:50.633840Z","shell.execute_reply.started":"2024-05-20T11:26:49.691581Z","shell.execute_reply":"2024-05-20T11:26:50.632865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!du -sh ../input/deepfake-detection-challenge/\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T11:26:50.636161Z","iopub.execute_input":"2024-05-20T11:26:50.636510Z","iopub.status.idle":"2024-05-20T11:26:52.885182Z","shell.execute_reply.started":"2024-05-20T11:26:50.636448Z","shell.execute_reply":"2024-05-20T11:26:52.884318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\n# Load metadata\ntrain_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\n\n# Display the first few rows of the metadata\nprint(train_sample_metadata.head())\n\n# Visualize the distribution of fake vs. real videos\nplt.figure(figsize=(8, 6))\nsns.countplot(data=train_sample_metadata, x='label')\nplt.title('Distribution of Fake vs. Real Videos')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n\n# Define a function to display a frame from a video\ndef display_video_frame(video_path):\n    cap = cv2.VideoCapture(video_path)\n    ret, frame = cap.read()\n    cap.release()\n    \n    if ret:\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        plt.imshow(frame)\n        plt.axis('off')\n        plt.show()\n    else:\n        print(f\"Failed to read video: {video_path}\")\n\n# Path to the video folder\nvideo_folder = '../input/deepfake-detection-challenge/train_sample_videos/'\n\n# Display some sample frames from fake videos\nfake_videos = train_sample_metadata[train_sample_metadata['label'] == 'FAKE'].index\nprint(\"Sample frames from Fake videos:\")\nfor video in fake_videos[:3]:  # Display the first 3 fake videos\n    print(f\"Video: {video}\")\n    display_video_frame(os.path.join(video_folder, video))\n\n# Display some sample frames from real videos\nreal_videos = train_sample_metadata[train_sample_metadata['label'] == 'REAL'].index\nprint(\"Sample frames from Real videos:\")\nfor video in real_videos[:3]:  # Display the first 3 real videos\n    print(f\"Video: {video}\")\n    display_video_frame(os.path.join(video_folder, video))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T13:21:17.905656Z","iopub.execute_input":"2024-05-20T13:21:17.906000Z","iopub.status.idle":"2024-05-20T13:21:19.938977Z","shell.execute_reply.started":"2024-05-20T13:21:17.905952Z","shell.execute_reply":"2024-05-20T13:21:19.938123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mtcnn tensorflow opencv-python pandas numpy scikit-learn\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom mtcnn import MTCNN\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom sklearn.model_selection import train_test_split\n\n# Initialize MTCNN face detector\ndetector = MTCNN()\n\n# Paths and setup\nvideo_folder = '../input/deepfake-detection-challenge/train_sample_videos/'\nmetadata_path = '../input/deepfake-detection-challenge/train_sample_videos/metadata.json'\n\n# Load metadata\ntrain_sample_metadata = pd.read_json(metadata_path).T\n\n# Split metadata into training and validation sets\ntrain_metadata, val_metadata = train_test_split(train_sample_metadata, test_size=0.2, random_state=42)\n\nclass VideoFrameGenerator(Sequence):\n    def __init__(self, metadata, batch_size=32, target_size=(224, 224), shuffle=True):\n        self.metadata = metadata\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.metadata))\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.metadata) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        batch_metadata = self.metadata.iloc[batch_indexes]\n        \n        X, y_labels = self.__data_generation(batch_metadata)\n        return X, y_labels\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n    \n    def __data_generation(self, batch_metadata):\n        X = []\n        y_labels = []\n        \n        for video_name, row in batch_metadata.iterrows():\n            video_path = os.path.join(video_folder, video_name)\n            label = 1 if row['label'] == 'FAKE' else 0\n            \n            cap = cv2.VideoCapture(video_path)\n            while cap.isOpened():\n                ret, frame = cap.read()\n                if not ret:\n                    break\n                \n                frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n                faces = detector.detect_faces(frame_rgb)\n                \n                for face in faces:\n                    x, y, width, height = face['box']\n                    face_img = frame_rgb[y:y+height, x:x+width]\n                    face_img = cv2.resize(face_img, self.target_size)  # Resize face to target_size\n                    face_array = img_to_array(face_img) / 255.0  # Normalize pixel values\n                    \n                    X.append(face_array)\n                    y_labels.append(label)\n                    \n                    if len(X) >= self.batch_size:\n                        cap.release()\n                        return np.array(X), np.array(y_labels)\n            \n            cap.release()\n        \n        # If we exit the loop and don't have enough samples, pad with the first few samples\n        while len(X) < self.batch_size:\n            X.append(X[0])\n            y_labels.append(y_labels[0])\n        \n        return np.array(X), np.array(y_labels)\n\n# Instantiate the generators\nbatch_size = 32\ntrain_generator = VideoFrameGenerator(train_metadata, batch_size=batch_size)\nval_generator = VideoFrameGenerator(val_metadata, batch_size=batch_size)\n\n# Build the model\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(\n    train_generator,\n    epochs=2,\n    validation_data=val_generator\n)\n\n# Evaluate the model \nloss, accuracy = model.evaluate(val_generator)\nprint(f\"Validation Loss: {loss}\")\nprint(f\"Validation Accuracy: {accuracy}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T11:31:53.512049Z","iopub.execute_input":"2024-05-20T11:31:53.512419Z","iopub.status.idle":"2024-05-20T12:44:39.160218Z","shell.execute_reply.started":"2024-05-20T11:31:53.512357Z","shell.execute_reply":"2024-05-20T12:44:39.159410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom mtcnn import MTCNN\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import img_to_array\n\n# Initialize MTCNN face detector\ndetector = MTCNN()\n\n# Load the trained model\n# model = load_model('path/to/your/trained_model.h5')  # Update with the actual path to your model\n\n# Function to detect and preprocess faces from a video\ndef extract_faces_from_video(video_path, target_size=(224, 224)):\n    cap = cv2.VideoCapture(video_path)\n    faces = []\n\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        detected_faces = detector.detect_faces(frame_rgb)\n\n        for face in detected_faces:\n            x, y, width, height = face['box']\n            face_img = frame_rgb[y:y+height, x:x+width]\n            face_img = cv2.resize(face_img, target_size)\n            face_array = img_to_array(face_img) / 255.0\n            faces.append(face_array)\n    \n    cap.release()\n    return np.array(faces)\n\n# Function to predict if the video is fake or real\ndef predict_video(video_path):\n    faces = extract_faces_from_video(video_path)\n    if len(faces) == 0:\n        print(\"No faces detected in the video.\")\n        return None\n\n    predictions = model.predict(faces)\n    avg_prediction = np.mean(predictions)\n\n    if avg_prediction > 0.5:\n        print(f\"The video '{video_path}' is predicted to be FAKE.\")\n    else:\n        print(f\"The video '{video_path}' is predicted to be REAL.\")\n\n    return avg_prediction\n\n# Test the prediction function with a sample video\nvideo_path = '/kaggle/input/deepfake-detection-challenge/test_videos/aassnaulhq.mp4'  # Update with the actual path to the test video\npredict_video(video_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T13:07:07.487553Z","iopub.execute_input":"2024-05-20T13:07:07.487875Z","iopub.status.idle":"2024-05-20T13:10:04.332166Z","shell.execute_reply.started":"2024-05-20T13:07:07.487829Z","shell.execute_reply":"2024-05-20T13:10:04.331191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}