{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8623401,"sourceType":"datasetVersion","datasetId":5162425}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n\ndef extract_zip(zip_path, extract_to='datasets/dfdc'):\n    os.makedirs(extract_to, exist_ok=True)\n    with zipfile.ZipFile(zip_path, 'r') as zip_ref:\n        zip_ref.extractall(extract_to)\n\n# Example usage\nzip_path = 'Celeb-DF.zip'  # Replace with the actual path to your zip file\nextract_to = 'datasets/dfdc'\nextract_zip(zip_path, extract_to)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef extract_frames(video_path, num_frames=5):\n    cap = cv2.VideoCapture(video_path)\n    if not cap.isOpened():\n        print(f\"Error: Could not open video {video_path}\")\n        return []\n    frames = []\n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    step = max(1, frame_count // num_frames)\n    \n    for i in range(0, frame_count, step):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, i)\n        ret, frame = cap.read()\n        if ret:\n            frames.append(frame)\n        else:\n            break\n    cap.release()\n    return frames\n\ndef detect_and_crop_faces(frames):\n    face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_frontalface_default.xml')\n    cropped_faces = []\n    \n    for frame in frames:\n        gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n        faces = face_cascade.detectMultiScale(gray, 1.1, 4)\n        for (x, y, w, h) in faces:\n            cropped_faces.append(gray[y:y+h, x:x+w])\n    \n    return cropped_faces\n\ndef save_cropped_faces(video_path, output_dir, num_frames=5):\n    os.makedirs(output_dir, exist_ok=True)\n    frames = extract_frames(video_path, num_frames)\n    faces = detect_and_crop_faces(frames)\n    \n    video_name = os.path.basename(video_path).split('.')[0]\n    for i, face in enumerate(faces):\n        face_path = os.path.join(output_dir, f\"{video_name}_face_{i}.jpg\")\n        cv2.imwrite(face_path, face)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:34:30.717884Z","iopub.execute_input":"2024-06-07T14:34:30.718282Z","iopub.status.idle":"2024-06-07T14:34:30.919715Z","shell.execute_reply.started":"2024-06-07T14:34:30.71825Z","shell.execute_reply":"2024-06-07T14:34:30.918758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n# Process all videos in real and synthesized directories\nbase_dir = '/kaggle/input/deepfake'\nreal_videos_dir = os.path.join(base_dir, 'Celeb-real')\nsynthesized_videos_dir = os.path.join(base_dir, 'Celeb-synthesis')\n\nreal_output_dir = os.path.join('/kaggle/working/', 'real_faces')\nsynthesized_output_dir = os.path.join(\"/kaggle/working/\", 'synthesized_faces')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:34:35.47449Z","iopub.execute_input":"2024-06-07T14:34:35.474921Z","iopub.status.idle":"2024-06-07T14:34:35.48161Z","shell.execute_reply.started":"2024-06-07T14:34:35.47489Z","shell.execute_reply":"2024-06-07T14:34:35.480465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfor video_file in os.listdir(real_videos_dir):\n    video_path = os.path.join(real_videos_dir, video_file)\n    save_cropped_faces(video_path, real_output_dir)\n\nfor video_file in os.listdir(synthesized_videos_dir):\n    video_path = os.path.join(synthesized_videos_dir, video_file)\n    save_cropped_faces(video_path, synthesized_output_dir)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:38:05.058208Z","iopub.execute_input":"2024-06-07T14:38:05.059365Z","iopub.status.idle":"2024-06-07T14:48:20.088328Z","shell.execute_reply.started":"2024-06-07T14:38:05.05932Z","shell.execute_reply":"2024-06-07T14:48:20.087192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport random\ndef list_image_files(directory):\n    image_files = []\n    for root, _, files in os.walk(directory):\n        for file in files:\n            if file.endswith('.jpg'):\n                image_files.append(os.path.join(root, file))\n    return image_files\n\ndef load_images_and_labels(image_paths, label):\n    images = []\n    labels = []\n    for image_path in image_paths:\n        image = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n        images.append(image)\n        labels.append(label)\n    return images, labels\n\n# Directories for real and fake images\nreal_images_dir = '/kaggle/working/real_faces'\nfake_images_dir = '/kaggle/working/synthesized_faces'\n\nreal_images = list_image_files(real_images_dir)\nfake_images = list_image_files(fake_images_dir)\n\"\"\"\nreal_images, real_labels = load_images_and_labels(real_images, 0)\nfake_images, fake_labels = load_images_and_labels(fake_images, 1)\n\nimages = real_images + fake_images\nlabels = real_labels + fake_labels\n\nX_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.15, random_state=42)\"\"\"\nreal_images = random.sample(real_images, len(real_images) // 4)\nfake_images = random.sample(fake_images, len(fake_images) // 4)\n\nreal_images, real_labels = load_images_and_labels(real_images, 0)\nfake_images, fake_labels = load_images_and_labels(fake_images, 1)\n\nimages = real_images + fake_images\nlabels = real_labels + fake_labels\n\nX_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.15, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:36:50.57475Z","iopub.status.idle":"2024-06-07T14:36:50.575277Z","shell.execute_reply.started":"2024-06-07T14:36:50.575015Z","shell.execute_reply":"2024-06-07T14:36:50.575037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import resample\n\n# Combine images and labels\ndata = list(zip(images, labels))\n\n# Separate real and fake examples\nreal_data = [item for item in data if item[1] == 0]\nfake_data = [item for item in data if item[1] == 1]\n\n# Resample real data to match the number of fake data\nreal_data_resampled = resample(real_data, replace=True, n_samples=len(fake_data), random_state=42)\n\n# Combine resampled real data with fake data\nresampled_data = real_data_resampled + fake_data\nrandom.shuffle(resampled_data)\n\n# Unzip the data into images and labels\nimages_resampled, labels_resampled = zip(*resampled_data)\n\n# Split the resampled data into training and testing sets\nX_train_resampled, X_test_resampled, y_train_resampled, y_test_resampled = train_test_split(images_resampled, labels_resampled, test_size=0.15, random_state=42)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications import ResNet50\nfrom keras.applications.resnet import preprocess_input\n\n# Define ResNet50 model for feature extraction\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\ndef extract_features_resnet(images):\n    features = []\n    for img in images:\n        img = cv2.resize(img, (224, 224))\n        img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)\n        img = preprocess_input(img)\n        img = np.expand_dims(img, axis=0)\n        feature = base_model.predict(img)\n        features.append(feature.flatten())\n    return features\n\n# Extract features from images\ntrain_features = extract_features_resnet(X_train)\ntest_features = extract_features_resnet(X_test)\n","metadata":{"_kg_hide-output":true,"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_features)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:12:07.067196Z","iopub.execute_input":"2024-06-07T06:12:07.068041Z","iopub.status.idle":"2024-06-07T06:12:07.076618Z","shell.execute_reply.started":"2024-06-07T06:12:07.068002Z","shell.execute_reply":"2024-06-07T06:12:07.075267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_features)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:12:07.758554Z","iopub.execute_input":"2024-06-07T06:12:07.758962Z","iopub.status.idle":"2024-06-07T06:12:07.766202Z","shell.execute_reply.started":"2024-06-07T06:12:07.758929Z","shell.execute_reply":"2024-06-07T06:12:07.76503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:13:41.596747Z","iopub.execute_input":"2024-06-07T06:13:41.597186Z","iopub.status.idle":"2024-06-07T06:13:41.604322Z","shell.execute_reply.started":"2024-06-07T06:13:41.597151Z","shell.execute_reply":"2024-06-07T06:13:41.603083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import LSTM, Dense\n\n# Define LSTM model for classification\nmodel = Sequential()\nmodel.add(LSTM(128, input_shape=(train_features[0].shape[0], 1)))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Reshape features for LSTM input\ntrain_features = np.array(train_features).reshape(-1, train_features[0].shape[0], 1)\ntest_features = np.array(test_features).reshape(-1, test_features[0].shape[0], 1)\n\n# Train the model\nmodel.fit(train_features, np.array(y_train), epochs=10, batch_size=32, validation_split=0.1)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:13:44.770944Z","iopub.execute_input":"2024-06-07T06:13:44.771341Z","iopub.status.idle":"2024-06-07T12:19:24.194461Z","shell.execute_reply.started":"2024-06-07T06:13:44.771309Z","shell.execute_reply":"2024-06-07T12:19:24.192079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, accuracy_score, classification_report\n\n# Evaluate the model\ny_pred = model.predict(test_features)\ny_pred = (y_pred > 0.5).astype(int)\n\nprint(f\"Accuracy: {accuracy_score(y_test, y_pred)}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T12:19:24.375396Z","iopub.execute_input":"2024-06-07T12:19:24.375807Z","iopub.status.idle":"2024-06-07T12:23:58.995904Z","shell.execute_reply.started":"2024-06-07T12:19:24.375767Z","shell.execute_reply":"2024-06-07T12:23:58.994752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:13:55.121987Z","iopub.execute_input":"2024-06-07T14:13:55.12245Z","iopub.status.idle":"2024-06-07T14:13:55.513973Z","shell.execute_reply.started":"2024-06-07T14:13:55.122411Z","shell.execute_reply":"2024-06-07T14:13:55.512717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2024-06-07T12:24:55.519432Z","iopub.execute_input":"2024-06-07T12:24:55.519848Z","iopub.status.idle":"2024-06-07T12:24:55.531171Z","shell.execute_reply.started":"2024-06-07T12:24:55.519816Z","shell.execute_reply":"2024-06-07T12:24:55.530088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('deepfake_detection_model.h5')\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\n\ndef predict_video(video_path, model_path='deepfake_detection_model.h5', detector_name='ORB'):\n    # Load the trained model\n    model = load_model(model_path)\n    \n    # Preprocess the video\n    save_cropped_faces(video_path, output_dir='temp_faces', num_frames=5)\n    image_paths = list_image_files('temp_faces')\n    \n    # Load and preprocess images\n    images, _ = load_images_and_labels(image_paths, 0)  # Label is dummy here\n    features = extract_features_resnet(images)\n    features = np.array(features).reshape(-1, features[0].shape[0], 1)\n    \n    # Predict\n    prediction = model.predict(features)\n    return 'FAKE' if np.mean(prediction) > 0.5 else 'REAL'\n\n# Example usage\nvideo_path = 'path/to/new_video.mp4'  # Replace with actual path to new video\nresult = predict_video(video_path)\nprint(f\"The video is predicted to be: {result}\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample, compute_class_weight\nimport random\nfrom keras.applications import ResNet50\nfrom keras.applications.resnet import preprocess_input\nfrom keras.models import Sequential, load_model\nfrom keras.layers import LSTM, Dense\n\n# Function to list image files\ndef list_image_files(directory):\n    image_files = []\n    for root, _, files in os.walk(directory):\n        for file in files:\n            if file.endswith('.jpg'):\n                image_files.append(os.path.join(root, file))\n    return image_files\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:48:43.101319Z","iopub.execute_input":"2024-06-07T14:48:43.102269Z","iopub.status.idle":"2024-06-07T14:48:57.846897Z","shell.execute_reply.started":"2024-06-07T14:48:43.10223Z","shell.execute_reply":"2024-06-07T14:48:57.84591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Function to load images and labels\ndef load_images_and_labels(image_paths, label):\n    images = []\n    labels = []\n    for image_path in image_paths:\n        image = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n        images.append(image)\n        labels.append(label)\n    return images, labels\n\n# Directories for real and fake images\nreal_images_dir = '/kaggle/working/real_faces'\nfake_images_dir = '/kaggle/working/synthesized_faces'\n\n# Load and resample images\nreal_images = list_image_files(real_images_dir)\nfake_images = list_image_files(fake_images_dir)\n\n# Resample to balance classes\nreal_images = random.sample(real_images, len(real_images) // 4)\nfake_images = random.sample(fake_images, len(fake_images) // 4)\n\nreal_images, real_labels = load_images_and_labels(real_images, 0)\nfake_images, fake_labels = load_images_and_labels(fake_images, 1)\n\nimages = real_images + fake_images\nlabels = real_labels + fake_labels\n\n# Combine and resample data\ndata = list(zip(images, labels))\nreal_data = [item for item in data if item[1] == 0]\nfake_data = [item for item in data if item[1] == 1]\nreal_data_resampled = resample(real_data, replace=True, n_samples=len(fake_data), random_state=42)\nresampled_data = real_data_resampled + fake_data\nrandom.shuffle(resampled_data)\nimages_resampled, labels_resampled = zip(*resampled_data)\n\n# Split data into training and testing sets\nX_train_resampled, X_test_resampled, y_train_resampled, y_test_resampled = train_test_split(images_resampled, labels_resampled, test_size=0.15, random_state=42)\n\n# ResNet50 feature extraction\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\ndef extract_features_resnet(images):\n    features = []\n    for img in images:\n        img = cv2.resize(img, (224, 224))\n        img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)\n        img = preprocess_input(img)\n        img = np.expand_dims(img, axis=0)\n        feature = base_model.predict(img)\n        features.append(feature.flatten())\n    return features\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T14:52:11.37728Z","iopub.execute_input":"2024-06-07T14:52:11.377708Z","iopub.status.idle":"2024-06-07T14:52:13.044641Z","shell.execute_reply.started":"2024-06-07T14:52:11.377675Z","shell.execute_reply":"2024-06-07T14:52:13.043457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features = extract_features_resnet(X_train_resampled)\ntest_features = extract_features_resnet(X_test_resampled)\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-06-07T14:52:17.709726Z","iopub.execute_input":"2024-06-07T14:52:17.710102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LSTM model\nmodel = Sequential()\nmodel.add(LSTM(128, input_shape=(train_features[0].shape[0], 1)))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\ntrain_features = np.array(train_features).reshape(-1, train_features[0].shape[0], 1)\ntest_features = np.array(test_features).reshape(-1, test_features[0].shape[0], 1)\n\n# Compute class weights\nclass_weights = compute_class_weight('balanced', classes=np.unique(labels), y=labels)\nclass_weight_dict = {i: class_weights[i] for i in range(len(class_weights))}\n\n# Train the model with class weights\nmodel.fit(train_features, np.array(y_train_resampled), epochs=10, batch_size=32, validation_split=0.1, class_weight=class_weight_dict)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T15:22:54.509296Z","iopub.execute_input":"2024-06-07T15:22:54.510147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Evaluate the model\ny_pred = model.predict(test_features)\ny_pred = (y_pred > 0.5).astype(int)\n\nprint(f\"Accuracy: {accuracy_score(y_test_resampled, y_pred)}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test_resampled, y_pred))\nprint(\"Classification Report:\")\nprint(classification_report(y_test_resampled, y_pred))\n\n# Save the trained model\nmodel.save('deepfake_detection_model.h5')\n\n# Prediction function\ndef predict_video(video_path, model_path='deepfake_detection_model.h5', detector_name='ORB'):\n    model = load_model(model_path)\n    save_cropped_faces(video_path, output_dir='temp_faces', num_frames=5)\n    image_paths = list_image_files('temp_faces')\n    images, _ = load_images_and_labels(image_paths, 0)\n    features = extract_features_resnet(images)\n    features = np.array(features).reshape(-1, features[0].shape[0], 1)\n    prediction = model.predict(features)\n    return 'FAKE' if np.mean(prediction) > 0.5 else 'REAL'\n\n# Example usage\nvideo_path = 'path/to/new_video.mp4'\nresult = predict_video(video_path)\nprint(f\"The video is predicted to be: {result}\")","metadata":{},"execution_count":null,"outputs":[]}]}