{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, GlobalAveragePooling2D\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-06-07T17:23:35.686999Z","iopub.execute_input":"2024-06-07T17:23:35.687348Z","iopub.status.idle":"2024-06-07T17:23:45.196110Z","shell.execute_reply.started":"2024-06-07T17:23:35.687296Z","shell.execute_reply":"2024-06-07T17:23:45.195177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, GlobalAveragePooling2D\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\n# Function to extract frames from video\ndef extract_frames(video_path, output_folder, max_frames=100):\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n    cap = cv2.VideoCapture(video_path)\n    count = 0\n    while cap.isOpened() and count < max_frames:\n        ret, frame = cap.read()\n        if not ret:\n            break\n        frame_path = os.path.join(output_folder, f\"frame_{count}.jpg\")\n        cv2.imwrite(frame_path, frame)\n        count += 1\n    cap.release()\n\n# Paths to the dataset\ntrain_video_dir = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ntest_video_dir = '/kaggle/input/deepfake-detection-challenge/test_videos'\noutput_frame_dir = '/kaggle/working/frames'","metadata":{"execution":{"iopub.status.busy":"2024-06-07T17:25:18.311022Z","iopub.execute_input":"2024-06-07T17:25:18.311445Z","iopub.status.idle":"2024-06-07T17:25:18.322757Z","shell.execute_reply.started":"2024-06-07T17:25:18.311353Z","shell.execute_reply":"2024-06-07T17:25:18.321594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_path = '/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json'\nmetadata = pd.read_json(metadata_path).T","metadata":{"execution":{"iopub.status.busy":"2024-06-07T17:25:33.474493Z","iopub.execute_input":"2024-06-07T17:25:33.474884Z","iopub.status.idle":"2024-06-07T17:25:33.850403Z","shell.execute_reply.started":"2024-06-07T17:25:33.474811Z","shell.execute_reply":"2024-06-07T17:25:33.848839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in tqdm(metadata.index):\n    label = metadata.loc[video_file, 'label']\n    video_path = os.path.join(train_video_dir, video_file)\n    output_folder = os.path.join(output_frame_dir, label, os.path.splitext(video_file)[0])\n    extract_frames(video_path, output_folder)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T17:25:36.241804Z","iopub.execute_input":"2024-06-07T17:25:36.242209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_gen = ImageDataGenerator(rescale=1./255, validation_split=0.2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = data_gen.flow_from_directory(\n    output_frame_dir,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='training'\n)\n\nvalidation_generator = data_gen.flow_from_directory(\n    output_frame_dir,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='validation'\n)\n\n# Build the model using ResNet50\nbase_model = ResNet50(include_top=False, input_shape=(224, 224, 3), weights='imagenet')\nbase_model.trainable = False\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = Sequential([\n    base_model,\n    GlobalAveragePooling2D(),\n    Dense(256, activation='relu'),\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    validation_data=validation_generator,\n    epochs=10\n)\n\n# Evaluate the model\nloss, accuracy = model.evaluate(validation_generator)\nprint(f\"Validation accuracy: {accuracy:.2f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}