{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":865847,"sourceType":"datasetVersion","datasetId":459871}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-15T11:23:00.996769Z","iopub.execute_input":"2024-10-15T11:23:00.99703Z","iopub.status.idle":"2024-10-15T11:23:01.27418Z","shell.execute_reply.started":"2024-10-15T11:23:00.996988Z","shell.execute_reply":"2024-10-15T11:23:01.273315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pylab as plt\nimport cv2\nplt.style.use('ggplot')\nfrom IPython.display import Video\nfrom IPython.display import HTML","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:01.276472Z","iopub.execute_input":"2024-10-15T11:23:01.276741Z","iopub.status.idle":"2024-10-15T11:23:01.302107Z","shell.execute_reply.started":"2024-10-15T11:23:01.276697Z","shell.execute_reply":"2024-10-15T11:23:01.301394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -GFlash ../input/deepfake-detection-challenge\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:01.30325Z","iopub.execute_input":"2024-10-15T11:23:01.303466Z","iopub.status.idle":"2024-10-15T11:23:02.308112Z","shell.execute_reply.started":"2024-10-15T11:23:01.303428Z","shell.execute_reply":"2024-10-15T11:23:02.307237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!du -sh ../input/deepfake-detection-challenge/\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:02.309599Z","iopub.execute_input":"2024-10-15T11:23:02.309846Z","iopub.status.idle":"2024-10-15T11:23:03.601389Z","shell.execute_reply.started":"2024-10-15T11:23:02.309804Z","shell.execute_reply":"2024-10-15T11:23:03.600648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:03.604502Z","iopub.execute_input":"2024-10-15T11:23:03.604777Z","iopub.status.idle":"2024-10-15T11:23:04.021906Z","shell.execute_reply.started":"2024-10-15T11:23:03.60473Z","shell.execute_reply":"2024-10-15T11:23:04.021001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\ntrain_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\nprint(train_sample_metadata.head())\nplt.figure(figsize=(8, 6))\nsns.countplot(data=train_sample_metadata, x='label')\nplt.title('Distribution of Fake vs. Real Videos')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n\ndef display_video_frame(video_path):\n    cap = cv2.VideoCapture(video_path)\n    ret, frame = cap.read()\n    cap.release()\n    \n    if ret:\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        plt.imshow(frame)\n        plt.axis('off')\n        plt.show()\n    else:\n        print(f\"Failed to read video: {video_path}\")\n\nvideo_folder = '../input/deepfake-detection-challenge/train_sample_videos/'\nfake_videos = train_sample_metadata[train_sample_metadata['label'] == 'FAKE'].index\nprint(\"Sample frames from Fake videos:\")\nfor video in fake_videos[:3]:  \n    print(f\"Video: {video}\")\n    display_video_frame(os.path.join(video_folder, video))\nreal_videos = train_sample_metadata[train_sample_metadata['label'] == 'REAL'].index\nprint(\"Sample frames from Real videos:\")\nfor video in real_videos[:3]:  # Display the first 3 real videos\n    print(f\"Video: {video}\")\n    display_video_frame(os.path.join(video_folder, video))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:04.023621Z","iopub.execute_input":"2024-10-15T11:23:04.023957Z","iopub.status.idle":"2024-10-15T11:23:07.077399Z","shell.execute_reply.started":"2024-10-15T11:23:04.023896Z","shell.execute_reply":"2024-10-15T11:23:07.07642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mtcnn tensorflow opencv-python pandas numpy scikit-learn\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:07.079611Z","iopub.execute_input":"2024-10-15T11:23:07.079959Z","iopub.status.idle":"2024-10-15T11:23:13.259996Z","shell.execute_reply.started":"2024-10-15T11:23:07.079898Z","shell.execute_reply":"2024-10-15T11:23:13.259078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom mtcnn import MTCNN\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom sklearn.model_selection import train_test_split\n\n# Initialize MTCNN face detector\ndetector = MTCNN()\nvideo_folder = '../input/deepfake-detection-challenge/train_sample_videos/'\nmetadata_path = '../input/deepfake-detection-challenge/train_sample_videos/metadata.json'\ntrain_sample_metadata = pd.read_json(metadata_path).T\n\ntrain_metadata, val_metadata = train_test_split(train_sample_metadata, test_size=0.2, random_state=42)\n\nclass VideoFrameGenerator(Sequence):\n    def __init__(self, metadata, batch_size=32, target_size=(224, 224), shuffle=True):\n        self.metadata = metadata\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.metadata))\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.metadata) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        batch_metadata = self.metadata.iloc[batch_indexes]\n        \n        X, y_labels = self.__data_generation(batch_metadata)\n        return X, y_labels\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n    \n    def __data_generation(self, batch_metadata):\n        X = []\n        y_labels = []\n        \n        for video_name, row in batch_metadata.iterrows():\n            video_path = os.path.join(video_folder, video_name)\n            label = 1 if row['label'] == 'FAKE' else 0\n            \n            cap = cv2.VideoCapture(video_path)\n            while cap.isOpened():\n                ret, frame = cap.read()\n                if not ret:\n                    break\n                \n                frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n                faces = detector.detect_faces(frame_rgb)\n                \n                for face in faces:\n                    x, y, width, height = face['box']\n                    face_img = frame_rgb[y:y+height, x:x+width]\n                    face_img = cv2.resize(face_img, self.target_size)  \n                    face_array = img_to_array(face_img) / 255.0  \n                    \n                    X.append(face_array)\n                    y_labels.append(label)\n                    \n                    if len(X) >= self.batch_size:\n                        cap.release()\n                        return np.array(X), np.array(y_labels)\n            \n            cap.release()\n        \n        while len(X) < self.batch_size:\n            X.append(X[0])\n            y_labels.append(y_labels[0])\n        \n        return np.array(X), np.array(y_labels)\n\nbatch_size = 32\ntrain_generator = VideoFrameGenerator(train_metadata, batch_size=batch_size)\nval_generator = VideoFrameGenerator(val_metadata, batch_size=batch_size)\n\n# model building\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(224, 224, 3)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(\n    train_generator,\n    epochs=2,\n    validation_data=val_generator\n)\n\nloss, accuracy = model.evaluate(val_generator)\nprint(f\"Validation Loss: {loss}\")\nprint(f\"Validation Accuracy: {accuracy}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:23:13.261948Z","iopub.execute_input":"2024-10-15T11:23:13.262241Z","iopub.status.idle":"2024-10-15T11:32:43.464925Z","shell.execute_reply.started":"2024-10-15T11:23:13.262186Z","shell.execute_reply":"2024-10-15T11:32:43.463705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    epochs=2,\n    validation_data=val_generator\n)\nloss, accuracy = model.evaluate(val_generator)\nprint(f\"Validation Loss: {loss}\")\nprint(f\"Validation Accuracy: {accuracy}\")\n\nmodel_save_path = '/kaggle/working/deepfake_detection_model.h5'  \nmodel.save(model_save_path)\nprint(f\"Model saved to {model_save_path}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:32:43.466605Z","iopub.execute_input":"2024-10-15T11:32:43.466946Z","iopub.status.idle":"2024-10-15T11:41:58.995719Z","shell.execute_reply.started":"2024-10-15T11:32:43.466882Z","shell.execute_reply":"2024-10-15T11:41:58.994377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom mtcnn import MTCNN\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import img_to_array\n\n# Initialize MTCNN face detector\ndetector = MTCNN()\n\ndef extract_faces_from_video(video_path, target_size=(224, 224)):\n    cap = cv2.VideoCapture(video_path)\n    faces = []\n\n    while cap.isOpened():\n        ret, frame = cap.read()\n        if not ret:\n            break\n\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        detected_faces = detector.detect_faces(frame_rgb)\n\n        for face in detected_faces:\n            x, y, width, height = face['box']\n            face_img = frame_rgb[y:y+height, x:x+width]\n            face_img = cv2.resize(face_img, target_size)\n            face_array = img_to_array(face_img) / 255.0\n            faces.append(face_array)\n    \n    cap.release()\n    return np.array(faces)\n\ndef predict_video(video_path):\n    faces = extract_faces_from_video(video_path)\n    if len(faces) == 0:\n        print(\"No faces detected in the video.\")\n        return None\n\n    predictions = model.predict(faces)\n    avg_prediction = np.mean(predictions)\n\n    if avg_prediction > 0.5:\n        print(f\"The video '{video_path}' is predicted to be FAKE.\")\n    else:\n        print(f\"The video '{video_path}' is predicted to be REAL.\")\n\n    return avg_prediction\nvideo_path = '/kaggle/input/deepfake-detection-challenge/test_videos/hierggamuo.mp4' \npredict_video(video_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T11:41:58.997291Z","iopub.execute_input":"2024-10-15T11:41:58.997635Z","iopub.status.idle":"2024-10-15T11:44:55.467666Z","shell.execute_reply.started":"2024-10-15T11:41:58.997574Z","shell.execute_reply":"2024-10-15T11:44:55.466453Z"},"trusted":true},"execution_count":null,"outputs":[]}]}