{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":4570814,"sourceType":"datasetVersion","datasetId":2666698}],"dockerImageVersionId":30777,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json5\n\n# Define dataset paths\ndataset_path = '/kaggle/input/deepfake-face-mask-dataset-dffmd'\nfake_videos_path = os.path.join(dataset_path, 'Fake', 'Fake')\nreal_videos_path = os.path.join(dataset_path, 'Real', 'Real')\n\n# Load the metadata\nmetadata_path = os.path.join(dataset_path, 'DFFD_metadata.json')\nwith open(metadata_path, 'r') as f:\n    metadata = json5.load(f)\n\n# Explore metadata\n# print(\"Metadata keys:\", list(metadata.keys()))\nprint(\"Sample metadata:\", metadata[list(metadata.keys())[0]])\n\n# List videos in the fake and real folders\nfake_videos = os.listdir(fake_videos_path)\nreal_videos = os.listdir(real_videos_path)\n\nprint(f\"Number of fake videos: {len(fake_videos)}\")\nprint(f\"Number of real videos: {len(real_videos)}\")\nprint(\"Sample fake videos:\", fake_videos[:5])\nprint(\"Sample real videos:\", real_videos[:5])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:03:19.788396Z","iopub.execute_input":"2024-10-05T13:03:19.788921Z","iopub.status.idle":"2024-10-05T13:03:22.772627Z","shell.execute_reply.started":"2024-10-05T13:03:19.788884Z","shell.execute_reply":"2024-10-05T13:03:22.771779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.models import Sequential\nfrom sklearn.model_selection import train_test_split\nimport json5","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:37.186563Z","iopub.execute_input":"2024-10-05T13:16:37.186953Z","iopub.status.idle":"2024-10-05T13:16:37.191627Z","shell.execute_reply.started":"2024-10-05T13:16:37.186923Z","shell.execute_reply":"2024-10-05T13:16:37.190771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nexcept ValueError:\n    print('TPU not found. Running on default strategy.')\n    strategy = tf.distribute.get_strategy()","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:03:56.610071Z","iopub.execute_input":"2024-10-05T13:03:56.610538Z","iopub.status.idle":"2024-10-05T13:04:01.389906Z","shell.execute_reply.started":"2024-10-05T13:03:56.610503Z","shell.execute_reply":"2024-10-05T13:04:01.389110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define paths\ndataset_path = '/kaggle/input/deepfake-face-mask-dataset-dffmd'\nfake_videos_path = os.path.join(dataset_path, 'Fake', 'Fake')\nreal_videos_path = os.path.join(dataset_path, 'Real', 'Real')\nmetadata_path = os.path.join(dataset_path, 'DFFD_metadata.json')","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:41.486621Z","iopub.execute_input":"2024-10-05T13:16:41.487341Z","iopub.status.idle":"2024-10-05T13:16:41.491331Z","shell.execute_reply.started":"2024-10-05T13:16:41.487305Z","shell.execute_reply":"2024-10-05T13:16:41.490571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters\nIMG_SIZE = 128  # Resize frames to 128x128\nFRAMES_PER_VIDEO = 10  # Number of frames to sample per video\nLIMITED_VIDEOS = 100  # Number of videos to use per class","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:43.707147Z","iopub.execute_input":"2024-10-05T13:16:43.707727Z","iopub.status.idle":"2024-10-05T13:16:43.711800Z","shell.execute_reply.started":"2024-10-05T13:16:43.707688Z","shell.execute_reply":"2024-10-05T13:16:43.710832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load metadata\nprint(\"Loading metadata...\")\nwith open(metadata_path, 'r') as f:\n    metadata = json5.load(f)\nprint(\"Metadata loaded successfully.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:45.797261Z","iopub.execute_input":"2024-10-05T13:16:45.797672Z","iopub.status.idle":"2024-10-05T13:16:48.825576Z","shell.execute_reply.started":"2024-10-05T13:16:45.797635Z","shell.execute_reply":"2024-10-05T13:16:48.824585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to extract frames from video\ndef extract_frames(video_path, num_frames=FRAMES_PER_VIDEO):\n    print(f\"Extracting frames from video: {video_path}\")\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    print(f\"Total frames in video: {total_frames}\")\n    frame_indices = np.linspace(0, total_frames - 1, num_frames, dtype=int)\n    for idx in frame_indices:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n        ret, frame = cap.read()\n        if ret:\n            frame = cv2.resize(frame, (IMG_SIZE, IMG_SIZE))\n            frames.append(frame)\n    cap.release()\n    print(f\"Extracted {len(frames)} frames from video: {video_path}\")\n    return frames","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:50.940862Z","iopub.execute_input":"2024-10-05T13:16:50.941767Z","iopub.status.idle":"2024-10-05T13:16:50.947592Z","shell.execute_reply.started":"2024-10-05T13:16:50.941730Z","shell.execute_reply":"2024-10-05T13:16:50.946581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare data\nX, y = [], []","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:54.506500Z","iopub.execute_input":"2024-10-05T13:16:54.506977Z","iopub.status.idle":"2024-10-05T13:16:54.510934Z","shell.execute_reply.started":"2024-10-05T13:16:54.506938Z","shell.execute_reply":"2024-10-05T13:16:54.510116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process real videos\nprint(\"Processing real videos...\")\nreal_videos = os.listdir(real_videos_path)[:LIMITED_VIDEOS * 2]\nfor video_name in real_videos:\n    video_path = os.path.join(real_videos_path, video_name)\n    frames = extract_frames(video_path)\n    X.extend(frames)\n    y.extend([0] * len(frames))  # Label 0 for REAL\nprint(f\"Processed {len(real_videos)} real videos.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:16:57.368119Z","iopub.execute_input":"2024-10-05T13:16:57.369209Z","iopub.status.idle":"2024-10-05T13:20:17.522840Z","shell.execute_reply.started":"2024-10-05T13:16:57.369167Z","shell.execute_reply":"2024-10-05T13:20:17.521771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process fake videos\nprint(\"Processing fake videos...\")\nfake_videos = os.listdir(fake_videos_path)[:LIMITED_VIDEOS * 2]\nfor video_name in fake_videos:\n    video_path = os.path.join(fake_videos_path, video_name)\n    frames = extract_frames(video_path)\n    X.extend(frames)\n    y.extend([1] * len(frames))  # Label 1 for FAKE\nprint(f\"Processed {len(fake_videos)} fake videos.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:20:35.170364Z","iopub.execute_input":"2024-10-05T13:20:35.170804Z","iopub.status.idle":"2024-10-05T13:20:59.494491Z","shell.execute_reply.started":"2024-10-05T13:20:35.170772Z","shell.execute_reply":"2024-10-05T13:20:59.493498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert to numpy arrays\nprint(\"Converting data to numpy arrays...\")\nX = np.array(X) / 255.0  # Normalize pixel values\ny = np.array(y)\nprint(\"Data conversion complete. Shape of X: {}, Shape of y: {}\".format(X.shape, y.shape))","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:23:06.598974Z","iopub.execute_input":"2024-10-05T13:23:06.599404Z","iopub.status.idle":"2024-10-05T13:23:07.147486Z","shell.execute_reply.started":"2024-10-05T13:23:06.599372Z","shell.execute_reply":"2024-10-05T13:23:07.146413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into train and validation sets\nprint(\"Splitting data into train and validation sets...\")\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\nprint(\"Data split complete. Training samples: {}, Validation samples: {}\".format(len(y_train), len(y_val)))","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:23:19.184731Z","iopub.execute_input":"2024-10-05T13:23:19.185106Z","iopub.status.idle":"2024-10-05T13:23:19.645994Z","shell.execute_reply.started":"2024-10-05T13:23:19.185075Z","shell.execute_reply":"2024-10-05T13:23:19.645094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build a simple CNN model\nprint(\"Building CNN model...\")\nwith strategy.scope():\n    model = Sequential([\n        Conv2D(32, (3, 3), activation='relu', input_shape=(IMG_SIZE, IMG_SIZE, 3)),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Conv2D(128, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')\n    ])\nprint(\"Model built successfully.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:23:38.245668Z","iopub.execute_input":"2024-10-05T13:23:38.246050Z","iopub.status.idle":"2024-10-05T13:23:38.517072Z","shell.execute_reply.started":"2024-10-05T13:23:38.246019Z","shell.execute_reply":"2024-10-05T13:23:38.516196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model\nprint(\"Compiling the model...\")\nwith strategy.scope():\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nprint(\"Model compiled successfully.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:23:51.300629Z","iopub.execute_input":"2024-10-05T13:23:51.301043Z","iopub.status.idle":"2024-10-05T13:23:51.350329Z","shell.execute_reply.started":"2024-10-05T13:23:51.301008Z","shell.execute_reply":"2024-10-05T13:23:51.349389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Num TPU Available: \", len(tf.config.list_physical_devices('TPU')))","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:02:01.777736Z","iopub.execute_input":"2024-10-05T13:02:01.778668Z","iopub.status.idle":"2024-10-05T13:02:01.783146Z","shell.execute_reply.started":"2024-10-05T13:02:01.778628Z","shell.execute_reply":"2024-10-05T13:02:01.782367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nprint(\"Starting model training...\")\nhistory = model.fit(X_train, y_train, epochs=5, validation_data=(X_val, y_val), batch_size=16)\nprint(\"Model training complete.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:24:05.583385Z","iopub.execute_input":"2024-10-05T13:24:05.584433Z","iopub.status.idle":"2024-10-05T13:24:38.891949Z","shell.execute_reply.started":"2024-10-05T13:24:05.584362Z","shell.execute_reply":"2024-10-05T13:24:38.890952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nprint(\"Evaluating the model...\")\nval_loss, val_accuracy = model.evaluate(X_val, y_val)\nprint(f\"Validation Loss: {val_loss}\")\nprint(f\"Validation Accuracy: {val_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:24:43.698652Z","iopub.execute_input":"2024-10-05T13:24:43.699010Z","iopub.status.idle":"2024-10-05T13:24:46.890528Z","shell.execute_reply.started":"2024-10-05T13:24:43.698980Z","shell.execute_reply":"2024-10-05T13:24:46.889500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model\nwith strategy.scope():\n    model_path = '/kaggle/working/deepfake_detection_model.h5'\n    print(f\"Saving the model to {model_path}...\")\n    model.save(model_path)\n    print(\"Model saved successfully.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:26:28.997704Z","iopub.execute_input":"2024-10-05T13:26:28.998165Z","iopub.status.idle":"2024-10-05T13:26:29.181385Z","shell.execute_reply.started":"2024-10-05T13:26:28.998121Z","shell.execute_reply":"2024-10-05T13:26:29.180127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nprint(\"Evaluating the model...\")\nwith strategy.scope():\n    val_loss, val_accuracy = model.evaluate(X_val, y_val)\nprint(f\"Validation Loss: {val_loss}\")\nprint(f\"Validation Accuracy: {val_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:26:43.640703Z","iopub.execute_input":"2024-10-05T13:26:43.641058Z","iopub.status.idle":"2024-10-05T13:26:45.923852Z","shell.execute_reply.started":"2024-10-05T13:26:43.641029Z","shell.execute_reply":"2024-10-05T13:26:45.922852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test the model on a new video\ndef predict_video(video_path):\n    print(f\"Predicting on video: {video_path}\")\n    frames = extract_frames(video_path)\n    frames = np.array(frames) / 255.0\n    if frames.size == 0:\n        print(\"No frames extracted from video, skipping prediction.\")\n        return\n    # Predict frame by frame\n    predictions = []\n    for frame in frames:\n        frame = np.expand_dims(frame, axis=0)  # Add batch dimension for a single frame\n        prediction = model.predict(frame)\n        predictions.append(prediction[0][0])\n    \n    avg_prediction = np.mean(predictions)\n    label = \"FAKE\" if avg_prediction > 0.5 else \"REAL\"\n    print(f\"Average Prediction: {avg_prediction}, Label: {label}\")\n    return label","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:33:17.309185Z","iopub.execute_input":"2024-10-05T13:33:17.309652Z","iopub.status.idle":"2024-10-05T13:33:17.316639Z","shell.execute_reply.started":"2024-10-05T13:33:17.309616Z","shell.execute_reply":"2024-10-05T13:33:17.315133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage of prediction\nsample_video_path = os.path.join(fake_videos_path, fake_videos[0])\npredict_video(sample_video_path)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:33:19.516660Z","iopub.execute_input":"2024-10-05T13:33:19.517011Z","iopub.status.idle":"2024-10-05T13:33:25.886515Z","shell.execute_reply.started":"2024-10-05T13:33:19.516983Z","shell.execute_reply":"2024-10-05T13:33:25.885502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage of prediction\nsample_video_path = os.path.join(real_videos_path, real_videos[0])\npredict_video(sample_video_path)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T13:33:30.634037Z","iopub.execute_input":"2024-10-05T13:33:30.634409Z","iopub.status.idle":"2024-10-05T13:33:36.596039Z","shell.execute_reply.started":"2024-10-05T13:33:30.634378Z","shell.execute_reply":"2024-10-05T13:33:36.595070Z"},"trusted":true},"execution_count":null,"outputs":[]}]}