{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dlib\nimport cv2\nimport os\nimport re\nimport json\nfrom pylab import *\nfrom PIL import Image, ImageChops, ImageEnhance","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:41:48.036504Z","iopub.execute_input":"2024-06-08T09:41:48.036805Z","iopub.status.idle":"2024-06-08T09:41:48.041887Z","shell.execute_reply.started":"2024-06-08T09:41:48.036776Z","shell.execute_reply":"2024-06-08T09:41:48.040892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport dlib\n\ntrain_frame_folder = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\noutput_real_dir = '/kaggle/working/dataset/real'\noutput_fake_dir = '/kaggle/working/dataset/fake'\n\n# Create output directories if they don't exist\nos.makedirs(output_real_dir, exist_ok=True)\nos.makedirs(output_fake_dir, exist_ok=True)\n\n# Load metadata\nwith open(os.path.join(train_frame_folder, 'metadata.json'), 'r') as file:\n    data = json.load(file)\n\nlist_of_train_data = [f for f in os.listdir(train_frame_folder) if f.endswith('.mp4')]\ndetector = dlib.get_frontal_face_detector()\n\nfor vid in list_of_train_data:\n    print(f\"Processing video: {vid}\")\n    count = 0\n    cap = cv2.VideoCapture(os.path.join(train_frame_folder, vid))\n    frameRate = cap.get(cv2.CAP_PROP_FPS)\n\n    while cap.isOpened():\n        frameId = cap.get(cv2.CAP_PROP_POS_FRAMES)\n        ret, frame = cap.read()\n        if not ret:\n            break\n\n        if frameId % int(frameRate) == 0:\n            face_rects, scores, idx = detector.run(frame, 0)\n            for i, d in enumerate(face_rects):\n                x1 = d.left()\n                y1 = d.top()\n                x2 = d.right()\n                y2 = d.bottom()\n                crop_img = frame[y1:y2, x1:x2]\n                resized_img = cv2.resize(crop_img, (128, 128))\n\n                if data[vid]['label'] == 'REAL':\n                    output_path = os.path.join(output_real_dir, f\"{vid.split('.')[0]}_{count}.png\")\n                elif data[vid]['label'] == 'FAKE':\n                    output_path = os.path.join(output_fake_dir, f\"{vid.split('.')[0]}_{count}.png\")\n                \n                cv2.imwrite(output_path, resized_img)\n                count += 1\n\n    cap.release()\n    print(f\"Finished processing video: {vid}. Frames saved: {count}\")\n\nprint(\"Processing complete.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T09:44:37.561347Z","iopub.execute_input":"2024-06-08T09:44:37.562317Z","iopub.status.idle":"2024-06-08T10:16:28.796397Z","shell.execute_reply.started":"2024-06-08T09:44:37.562277Z","shell.execute_reply":"2024-06-08T10:16:28.795175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sn\nimport pandas as pd\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, img_to_array, load_img\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:16:28.798614Z","iopub.execute_input":"2024-06-08T10:16:28.798971Z","iopub.status.idle":"2024-06-08T10:16:37.070033Z","shell.execute_reply.started":"2024-06-08T10:16:28.798924Z","shell.execute_reply":"2024-06-08T10:16:37.069005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_shape = (128, 128, 3)\ndata_dir = '/kaggle/working/dataset'\n\nreal_data = [f for f in os.listdir(data_dir+'/real') if f.endswith('.png')]\nfake_data = [f for f in os.listdir(data_dir+'/fake') if f.endswith('.png')]\n\nX = []\nY = []\n\nfor img in real_data:\n    X.append(img_to_array(load_img(data_dir+'/real/'+img)).flatten() / 255.0)\n    Y.append(1)\nfor img in fake_data:\n    X.append(img_to_array(load_img(data_dir+'/fake/'+img)).flatten() / 255.0)\n    Y.append(0)\n\nY_val_org = Y\n\n#Normalization\nX = np.array(X)\nY = to_categorical(Y, 2)\n\n#Reshape\nX = X.reshape(-1, 128, 128, 3)\n\n#Train-Test split\nX_train, X_val, Y_train, Y_val = train_test_split(X, Y, test_size = 0.2, random_state=5)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:16:48.018995Z","iopub.execute_input":"2024-06-08T10:16:48.019399Z","iopub.status.idle":"2024-06-08T10:16:52.886950Z","shell.execute_reply.started":"2024-06-08T10:16:48.019366Z","shell.execute_reply":"2024-06-08T10:16:52.886074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionResNetV2\nfrom tensorflow.keras.layers import Conv2D\nfrom tensorflow.keras.layers import MaxPooling2D\nfrom tensorflow.keras.layers import Flatten\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.layers import InputLayer\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping\n\ngoogleNet_model = InceptionResNetV2(include_top=False, weights='imagenet', input_shape=input_shape)\ngoogleNet_model.trainable = True\nmodel = Sequential()\nmodel.add(googleNet_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(units=2, activation='softmax'))\nmodel.compile(loss='binary_crossentropy',\n              optimizer=optimizers.Adam(lr=1e-5, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.0, amsgrad=False),\n              metrics=['accuracy'])\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:17:32.018816Z","iopub.execute_input":"2024-06-08T10:17:32.019225Z","iopub.status.idle":"2024-06-08T10:17:43.509274Z","shell.execute_reply.started":"2024-06-08T10:17:32.019190Z","shell.execute_reply":"2024-06-08T10:17:43.508268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Currently not used\nearly_stopping = EarlyStopping(monitor='val_loss',\n                               min_delta=0,\n                               patience=2,\n                               verbose=0, mode='auto')\nEPOCHS = 20\nBATCH_SIZE = 100\nhistory = model.fit(X_train, Y_train, batch_size = BATCH_SIZE, epochs = EPOCHS, validation_data = (X_val, Y_val), verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:19:15.739818Z","iopub.execute_input":"2024-06-08T10:19:15.740872Z","iopub.status.idle":"2024-06-08T10:25:55.815224Z","shell.execute_reply.started":"2024-06-08T10:19:15.740828Z","shell.execute_reply":"2024-06-08T10:25:55.813830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, (ax1, ax2) = plt.subplots(1, 2, figsize=(20, 4))\nt = f.suptitle('Pre-trained InceptionResNetV2 Transfer Learn with Fine-Tuning & Image Augmentation Performance ', fontsize=12)\nf.subplots_adjust(top=0.85, wspace=0.3)\n\nepoch_list = list(range(1,EPOCHS+1))\nax1.plot(epoch_list, history.history['accuracy'], label='Train Accuracy')\nax1.plot(epoch_list, history.history['val_accuracy'], label='Validation Accuracy')\nax1.set_xticks(np.arange(0, EPOCHS+1, 1))\nax1.set_ylabel('Accuracy Value')\nax1.set_xlabel('Epoch #')\nax1.set_title('Accuracy')\nl1 = ax1.legend(loc=\"best\")\n\nax2.plot(epoch_list, history.history['loss'], label='Train Loss')\nax2.plot(epoch_list, history.history['val_loss'], label='Validation Loss')\nax2.set_xticks(np.arange(0, EPOCHS+1, 1))\nax2.set_ylabel('Loss Value')\nax2.set_xlabel('Epoch #')\nax2.set_title('Loss')\nl2 = ax2.legend(loc=\"best\")","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:25:58.732706Z","iopub.execute_input":"2024-06-08T10:25:58.733137Z","iopub.status.idle":"2024-06-08T10:25:59.260234Z","shell.execute_reply.started":"2024-06-08T10:25:58.733099Z","shell.execute_reply":"2024-06-08T10:25:59.259269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sn\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix\n\n# Function to print confusion matrix\ndef print_confusion_matrix(y_true, y_pred_prob, threshold=0.5):\n    # Convert probabilities or one-hot encoded predictions to binary class labels\n    if y_pred_prob.ndim == 2 and y_pred_prob.shape[1] > 1:\n        y_pred_prob = y_pred_prob[:, 1]  # Select probabilities of class 1 if one-hot encoded\n    \n    y_pred = (y_pred_prob >= threshold).astype(int)\n    \n    cm = confusion_matrix(y_true, y_pred)\n    print('True positive = ', cm[0][0])\n    print('False positive = ', cm[0][1])\n    print('False negative = ', cm[1][0])\n    print('True negative = ', cm[1][1])\n    print('\\n')\n    \n    df_cm = pd.DataFrame(cm, range(2), range(2))\n    sn.set(font_scale=1.4) # for label size\n    sn.heatmap(df_cm, annot=True, annot_kws={\"size\": 16}) # font size\n    plt.ylabel('Actual label', size = 20)\n    plt.xlabel('Predicted label', size = 20)\n    plt.xticks(np.arange(2), ['Fake', 'Real'], size = 16)\n    plt.yticks(np.arange(2), ['Fake', 'Real'], size = 16)\n    plt.ylim([2, 0])\n    plt.show()\n\n# Assuming model.predict returns probabilities, apply the function\ny_pred_prob = model.predict(X)\nprint_confusion_matrix(Y_val_org, y_pred_prob)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:29:17.164674Z","iopub.execute_input":"2024-06-08T10:29:17.169630Z","iopub.status.idle":"2024-06-08T10:29:27.429375Z","shell.execute_reply.started":"2024-06-08T10:29:17.169580Z","shell.execute_reply":"2024-06-08T10:29:27.428377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('/kaggle/working/deepfake-detection-model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:41:53.321727Z","iopub.execute_input":"2024-06-08T10:41:53.322205Z","iopub.status.idle":"2024-06-08T10:41:56.184729Z","shell.execute_reply.started":"2024-06-08T10:41:53.322168Z","shell.execute_reply":"2024-06-08T10:41:56.183469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport dlib\nimport numpy as np\nfrom keras.preprocessing.image import img_to_array\n\ninput_shape = (128, 128, 3)\ndetector = dlib.get_frontal_face_detector()\ncap = cv2.VideoCapture('/kaggle/input/deepfake-detection-challenge/train_sample_videos/abarnvbtwb.mp4')\nframeRate = cap.get(5)\n\nwhile cap.isOpened():\n    frameId = cap.get(1)\n    ret, frame = cap.read()\n    if not ret:\n        break\n    if frameId % ((int(frameRate) + 1) * 1) == 0:\n        face_rects, scores, idx = detector.run(frame, 0)\n        for i, d in enumerate(face_rects):\n            x1 = d.left()\n            y1 = d.top()\n            x2 = d.right()\n            y2 = d.bottom()\n            crop_img = frame[y1:y2, x1:x2]\n            data = img_to_array(cv2.resize(crop_img, input_shape[:2])) / 255.0\n            data = np.expand_dims(data, axis=0)\n            predictions = model.predict(data)\n            predicted_class = np.argmax(predictions, axis=1)\n            print(predicted_class)\n\ncap.release()\ncv2.destroyAllWindows()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:32:48.987065Z","iopub.execute_input":"2024-06-08T10:32:48.987870Z","iopub.status.idle":"2024-06-08T10:32:55.303805Z","shell.execute_reply.started":"2024-06-08T10:32:48.987835Z","shell.execute_reply":"2024-06-08T10:32:55.302274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport dlib\nimport numpy as np\nfrom keras.preprocessing.image import img_to_array\n\n# Define the input shape for the model\ninput_shape = (128, 128, 3)\n\n# Initialize the face detector\ndetector = dlib.get_frontal_face_detector()\n\n# Open the video file\ncap = cv2.VideoCapture('/kaggle/input/deepfake-detection-challenge/train_sample_videos/abarnvbtwb.mp4')\n\n# Get the frame rate of the video\nframeRate = cap.get(5)\n\n# List to store the predictions\npredictions_summary = []\n\n# Process the video frame by frame\nwhile cap.isOpened():\n    frameId = cap.get(1)  # Current frame number\n    ret, frame = cap.read()  # Read the next frame\n    if not ret:\n        break  # Exit the loop if there are no more frames\n\n    # Process one frame every second\n    if frameId % ((int(frameRate) + 1) * 1) == 0:\n        # Detect faces in the frame\n        face_rects, scores, idx = detector.run(frame, 0)\n        for i, d in enumerate(face_rects):\n            # Get the coordinates of the bounding box\n            x1 = d.left()\n            y1 = d.top()\n            x2 = d.right()\n            y2 = d.bottom()\n            \n            # Crop and preprocess the face image\n            crop_img = frame[y1:y2, x1:x2]\n            data = img_to_array(cv2.resize(crop_img, input_shape[:2])) / 255.0\n            data = np.expand_dims(data, axis=0)  # Add batch dimension\n            \n            # Predict the class label\n            predictions = model.predict(data)\n            predicted_class = (predictions > 0.5).astype(int)  # Assuming binary classification with sigmoid activation\n            \n            # Determine if the prediction is Fake or Real\n            result = 'Fake' if predicted_class[0][0] == 1 else 'Real'\n            predictions_summary.append((frameId, result))\n\n\n\n# Print the summary of predictions\nfor frame_id, result in predictions_summary:\n    print(f'Frame {frame_id}: {result}')\n\n# Print final summary\nfake_count = sum(1 for _, result in predictions_summary if result == 'Fake')\nreal_count = sum(1 for _, result in predictions_summary if result == 'Real')\n\nprint(\"\\nFinal Summary:\")\nprint(f'Total Fake frames: {fake_count}')\nprint(f'Total Real frames: {real_count}')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:35:11.069159Z","iopub.execute_input":"2024-06-08T10:35:11.069559Z","iopub.status.idle":"2024-06-08T10:35:16.911569Z","shell.execute_reply.started":"2024-06-08T10:35:11.069520Z","shell.execute_reply":"2024-06-08T10:35:16.910511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport dlib\nimport numpy as np\nfrom keras.preprocessing.image import img_to_array\nfrom keras.models import load_model\n\n# Define the input shape for the model\ninput_shape = (128, 128, 3)\n\n# Initialize the face detector\ndetector = dlib.get_frontal_face_detector()\n\n# Open the video file\ncap = cv2.VideoCapture('/kaggle/input/deepfake-detection-challenge/train_sample_videos/aagfhgtpmv.mp4')\n\n# Get the frame rate of the video\nframeRate = cap.get(5)\n\n\n# List to store the predictions\npredictions_summary = []\n\n# Process the video frame by frame\nwhile cap.isOpened():\n    frameId = cap.get(1)  # Current frame number\n    ret, frame = cap.read()  # Read the next frame\n    if not ret:\n        break  # Exit the loop if there are no more frames\n\n    # Process one frame every second\n    if frameId % ((int(frameRate) + 1) * 1) == 0:\n        # Detect faces in the frame\n        face_rects, scores, idx = detector.run(frame, 0)\n        for i, d in enumerate(face_rects):\n            # Get the coordinates of the bounding box\n            x1 = d.left()\n            y1 = d.top()\n            x2 = d.right()\n            y2 = d.bottom()\n            \n            # Crop and preprocess the face image\n            crop_img = frame[y1:y2, x1:x2]\n            data = img_to_array(cv2.resize(crop_img, input_shape[:2])) / 255.0\n            data = np.expand_dims(data, axis=0)  # Add batch dimension\n            \n            # Predict the class label\n            predictions = model.predict(data)\n            predicted_class = (predictions > 0.5).astype(int)  # Assuming binary classification with sigmoid activation\n            \n            # Determine if the prediction is Fake or Real\n            result = 'Fake' if predicted_class[0][0] == 1 else 'Real'\n            predictions_summary.append(result)\n\n\n\n# Determine the overall classification of the video\nfake_count = predictions_summary.count('Fake')\nreal_count = predictions_summary.count('Real')\n\noverall_result = 'Fake' if fake_count > real_count else 'Real'\n\n# Print the final classification result\nprint(f'Final Classification of the video: {overall_result}')\nprint(f'Total Fake frames: {fake_count}')\nprint(f'Total Real frames: {real_count}')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-08T10:36:57.063836Z","iopub.execute_input":"2024-06-08T10:36:57.064724Z","iopub.status.idle":"2024-06-08T10:37:03.203705Z","shell.execute_reply.started":"2024-06-08T10:36:57.064685Z","shell.execute_reply":"2024-06-08T10:37:03.202247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}