{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5380830,"sourceType":"datasetVersion","datasetId":3120670}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-21T12:41:04.075418Z","iopub.execute_input":"2024-09-21T12:41:04.075818Z","iopub.status.idle":"2024-09-21T12:41:09.992664Z","shell.execute_reply.started":"2024-09-21T12:41:04.075737Z","shell.execute_reply":"2024-09-21T12:41:09.991923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ncelebd_df_v2_path = '../input/celeb-df-v2/' \nprint(os.listdir(celebd_df_v2_path))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T12:41:37.095845Z","iopub.execute_input":"2024-09-21T12:41:37.096148Z","iopub.status.idle":"2024-09-21T12:41:37.102199Z","shell.execute_reply.started":"2024-09-21T12:41:37.096106Z","shell.execute_reply":"2024-09-21T12:41:37.101333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ncelebd_df_v2_path = '../input/celeb-df-v2/'\n\nvideo_paths = []\nlabels = []\n\nyoutube_real_path = os.path.join(celebd_df_v2_path, 'YouTube-real')\nfor video_file in os.listdir(youtube_real_path):\n    video_paths.append(os.path.join(youtube_real_path, video_file))\n    labels.append('REAL')\n\nceleb_synthesis_path = os.path.join(celebd_df_v2_path, 'Celeb-synthesis')\nfor video_file in os.listdir(celeb_synthesis_path):\n    video_paths.append(os.path.join(celeb_synthesis_path, video_file))\n    labels.append('FAKE')\n\nceleb_real_path = os.path.join(celebd_df_v2_path, 'Celeb-real')\nfor video_file in os.listdir(celeb_real_path):\n    video_paths.append(os.path.join(celeb_real_path, video_file))\n    labels.append('REAL')\n\ndf = pd.DataFrame({'video_path': video_paths, 'label': labels})\n\nprint(df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T12:41:43.873181Z","iopub.execute_input":"2024-09-21T12:41:43.873485Z","iopub.status.idle":"2024-09-21T12:41:43.920233Z","shell.execute_reply.started":"2024-09-21T12:41:43.873439Z","shell.execute_reply":"2024-09-21T12:41:43.91943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\ndef extract_frames(video_path, num_frames=10):\n    frames = []\n    cap = cv2.VideoCapture(video_path)\n    \n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    frame_interval = total_frames // num_frames\n    \n    for i in range(num_frames):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, i * frame_interval)\n        ret, frame = cap.read()\n        if ret:\n            frames.append(frame)\n    \n    cap.release()\n    return frames\n\nsample_video = df.video_path[0] \nframes = extract_frames(sample_video)\n\nprint(len(frames))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T12:42:16.577813Z","iopub.execute_input":"2024-09-21T12:42:16.578111Z","iopub.status.idle":"2024-09-21T12:42:16.95891Z","shell.execute_reply.started":"2024-09-21T12:42:16.578067Z","shell.execute_reply":"2024-09-21T12:42:16.958021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef preprocess_frames(frames, target_size=(224, 224)):\n    processed_frames = []\n    for frame in frames:\n        frame = cv2.resize(frame, target_size)\n        frame = frame.astype('float32') / 255.0  # Normalize pixel values to [0, 1]\n        processed_frames.append(frame)\n    return np.array(processed_frames)\n\nprocessed_frames = preprocess_frames(frames)\nprint(processed_frames.shape)  # Should show (num_frames, height, width, channels)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T12:42:22.254192Z","iopub.execute_input":"2024-09-21T12:42:22.254496Z","iopub.status.idle":"2024-09-21T12:42:22.281823Z","shell.execute_reply.started":"2024-09-21T12:42:22.254449Z","shell.execute_reply":"2024-09-21T12:42:22.280913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\n\ndef create_resnet_model():\n    base_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    predictions = Dense(1, activation='sigmoid')(x)  # Binary classification\n    \n    model = Model(inputs=base_model.input, outputs=predictions)\n    \n    for layer in base_model.layers:\n        layer.trainable = False\n    \n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\nmodel = create_resnet_model()\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T12:42:37.453853Z","iopub.execute_input":"2024-09-21T12:42:37.454145Z","iopub.status.idle":"2024-09-21T12:42:48.632601Z","shell.execute_reply.started":"2024-09-21T12:42:37.454104Z","shell.execute_reply":"2024-09-21T12:42:48.631833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***REAL-1***","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\n\ndef extract_frames_from_videos(video_paths, target_folder, num_frames=10):\n    if not os.path.exists(target_folder):\n        os.makedirs(target_folder)\n\n    for video_path in video_paths:\n        video_name = os.path.basename(video_path).split('.')[0]\n        cap = cv2.VideoCapture(video_path)\n\n        total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        frame_interval = total_frames // num_frames\n        frames = []\n\n        for i in range(num_frames):\n            cap.set(cv2.CAP_PROP_POS_FRAMES, i * frame_interval)\n            ret, frame = cap.read()\n            if ret:\n                frame = cv2.resize(frame, (224, 224))\n                frames.append(frame)\n                frame_filename = os.path.join(target_folder, f\"{video_name}_frame_{i}.jpg\")\n                cv2.imwrite(frame_filename, frame)\n\n        cap.release()\n\nvideo_paths = df['video_path'].tolist()  # List of video paths from your DataFrame\nextract_frames_from_videos(video_paths, 'extracted_frames')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:02:27.933361Z","iopub.execute_input":"2024-09-21T13:02:27.933654Z","iopub.status.idle":"2024-09-21T13:13:43.551146Z","shell.execute_reply.started":"2024-09-21T13:02:27.9336Z","shell.execute_reply":"2024-09-21T13:13:43.550338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_paths = []\nlabels = []\n\nfor video_path in video_paths:\n    video_name = os.path.basename(video_path).split('.')[0]\n    for i in range(10):\n        frame_paths.append(f'extracted_frames/{video_name}_frame_{i}.jpg')\n        labels.append(df[df['video_path'] == video_path]['label'].values[0])\n\nframe_df = pd.DataFrame({'frame_path': frame_paths, 'label': labels})\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:16:18.264864Z","iopub.execute_input":"2024-09-21T13:16:18.265187Z","iopub.status.idle":"2024-09-21T13:18:04.938298Z","shell.execute_reply.started":"2024-09-21T13:16:18.265143Z","shell.execute_reply":"2024-09-21T13:18:04.93748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndatagen = ImageDataGenerator(validation_split=0.2)\n\ntrain_generator = datagen.flow_from_dataframe(\n    dataframe=frame_df,\n    x_col='frame_path',\n    y_col='label',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='training'\n)\n\nvalidation_generator = datagen.flow_from_dataframe(\n    dataframe=frame_df,\n    x_col='frame_path',\n    y_col='label',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='validation'\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfor path in frame_df['frame_path']:\n    if not os.path.exists(path):\n        print(f\"Missing frame: {path}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:19:04.973438Z","iopub.execute_input":"2024-09-21T13:19:04.973787Z","iopub.status.idle":"2024-09-21T13:19:05.248029Z","shell.execute_reply.started":"2024-09-21T13:19:04.973711Z","shell.execute_reply":"2024-09-21T13:19:05.24709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"videos_to_exclude = ['id27_0005']\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:19:07.76753Z","iopub.execute_input":"2024-09-21T13:19:07.767852Z","iopub.status.idle":"2024-09-21T13:19:07.771556Z","shell.execute_reply.started":"2024-09-21T13:19:07.767799Z","shell.execute_reply":"2024-09-21T13:19:07.77065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_frame_df = frame_df[~frame_df['frame_path'].str.contains('|'.join(videos_to_exclude))]\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:19:10.862417Z","iopub.execute_input":"2024-09-21T13:19:10.862721Z","iopub.status.idle":"2024-09-21T13:19:10.923468Z","shell.execute_reply.started":"2024-09-21T13:19:10.862678Z","shell.execute_reply":"2024-09-21T13:19:10.922884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = datagen.flow_from_dataframe(\n    dataframe=filtered_frame_df,\n    x_col='frame_path',\n    y_col='label',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='training'\n)\n\nvalidation_generator = datagen.flow_from_dataframe(\n    dataframe=filtered_frame_df,\n    x_col='frame_path',\n    y_col='label',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='binary',\n    subset='validation'\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:19:14.124823Z","iopub.execute_input":"2024-09-21T13:19:14.125193Z","iopub.status.idle":"2024-09-21T13:19:15.480639Z","shell.execute_reply.started":"2024-09-21T13:19:14.125124Z","shell.execute_reply":"2024-09-21T13:19:15.479881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout\nfrom tensorflow.keras.optimizers import Adam\n\nbase_model = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\nfor layer in base_model.layers:\n    layer.trainable = False\n\nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(optimizer=Adam(learning_rate=1e-4), loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-09-21T13:43:18.931608Z","iopub.execute_input":"2024-09-21T13:43:18.931909Z","iopub.status.idle":"2024-09-21T13:43:23.091919Z","shell.execute_reply.started":"2024-09-21T13:43:18.931866Z","shell.execute_reply":"2024-09-21T13:43:23.091043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nhistory = model.fit(\n    train_generator,\n    validation_data=validation_generator,\n    epochs=10,\n    steps_per_epoch=len(train_generator),\n    validation_steps=len(validation_generator)\n)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('deepfake_detection_model.h5')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}