{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":9146200,"sourceType":"datasetVersion","datasetId":5524489},{"sourceId":10195730,"sourceType":"datasetVersion","datasetId":6299820}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from tensorflow import keras\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:19.739775Z","iopub.execute_input":"2024-12-14T06:44:19.740071Z","iopub.status.idle":"2024-12-14T06:44:19.745037Z","shell.execute_reply.started":"2024-12-14T06:44:19.740020Z","shell.execute_reply":"2024-12-14T06:44:19.744079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '/kaggle/input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:19.746493Z","iopub.execute_input":"2024-12-14T06:44:19.746717Z","iopub.status.idle":"2024-12-14T06:44:19.764713Z","shell.execute_reply.started":"2024-12-14T06:44:19.746668Z","shell.execute_reply":"2024-12-14T06:44:19.763710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:19.766477Z","iopub.execute_input":"2024-12-14T06:44:19.766680Z","iopub.status.idle":"2024-12-14T06:44:19.888728Z","shell.execute_reply.started":"2024-12-14T06:44:19.766643Z","shell.execute_reply":"2024-12-14T06:44:19.888094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(5,5),kind='bar',title='The Label in the Training Set')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:19.890682Z","iopub.execute_input":"2024-12-14T06:44:19.890901Z","iopub.status.idle":"2024-12-14T06:44:20.094675Z","shell.execute_reply.started":"2024-12-14T06:44:19.890857Z","shell.execute_reply":"2024-12-14T06:44:20.093653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:20.096294Z","iopub.execute_input":"2024-12-14T06:44:20.096689Z","iopub.status.idle":"2024-12-14T06:44:20.104948Z","shell.execute_reply.started":"2024-12-14T06:44:20.096629Z","shell.execute_reply":"2024-12-14T06:44:20.103829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(5).index)\nf_train_sample_video","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:20.106687Z","iopub.execute_input":"2024-12-14T06:44:20.107277Z","iopub.status.idle":"2024-12-14T06:44:20.122353Z","shell.execute_reply.started":"2024-12-14T06:44:20.107212Z","shell.execute_reply":"2024-12-14T06:44:20.121089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def capture_image_from_video(video_path):\n    capture_image = cv2.VideoCapture(video_path)\n    ret, frame = capture_image.read()\n    fig = plt.figure(figsize =(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:20.124257Z","iopub.execute_input":"2024-12-14T06:44:20.124798Z","iopub.status.idle":"2024-12-14T06:44:20.134822Z","shell.execute_reply.started":"2024-12-14T06:44:20.124736Z","shell.execute_reply":"2024-12-14T06:44:20.133774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in f_train_sample_video:\n    capture_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:20.136570Z","iopub.execute_input":"2024-12-14T06:44:20.137134Z","iopub.status.idle":"2024-12-14T06:44:23.203765Z","shell.execute_reply.started":"2024-12-14T06:44:20.137063Z","shell.execute_reply":"2024-12-14T06:44:23.202992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(5).index)\nr_train_sample_video","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:23.205045Z","iopub.execute_input":"2024-12-14T06:44:23.205307Z","iopub.status.idle":"2024-12-14T06:44:23.212194Z","shell.execute_reply.started":"2024-12-14T06:44:23.205252Z","shell.execute_reply":"2024-12-14T06:44:23.211339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in r_train_sample_video:\n    capture_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:23.213147Z","iopub.execute_input":"2024-12-14T06:44:23.213338Z","iopub.status.idle":"2024-12-14T06:44:26.433790Z","shell.execute_reply.started":"2024-12-14T06:44:23.213292Z","shell.execute_reply":"2024-12-14T06:44:26.433134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f_videos = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:26.436704Z","iopub.execute_input":"2024-12-14T06:44:26.436899Z","iopub.status.idle":"2024-12-14T06:44:26.441720Z","shell.execute_reply.started":"2024-12-14T06:44:26.436864Z","shell.execute_reply":"2024-12-14T06:44:26.440933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\n\ndef play_video(video_file,subset=TRAIN_SAMPLE_FOLDER):\n    video_url = open(os.path.join(DATA_FOLDER,subset,video_file),'rb').read()\n    data_url = \"data:video/mp4;base64,\" + b64encode(video_url).decode()\n    return HTML(\"\"\"<video width=500 controls><source src=\"%s\" type=\"video/mp4\"></video>\"\"\" %data_url)\nplay_video(f_videos[5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:26.443086Z","iopub.execute_input":"2024-12-14T06:44:26.443267Z","iopub.status.idle":"2024-12-14T06:44:26.514826Z","shell.execute_reply.started":"2024-12-14T06:44:26.443234Z","shell.execute_reply":"2024-12-14T06:44:26.513352Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Modelling**","metadata":{}},{"cell_type":"code","source":"img_size = 224\nbatch_size = 64\nepochs = 15\n\nmax_seq_length = 20\nnum_features = 2048","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:26.516708Z","iopub.execute_input":"2024-12-14T06:44:26.517054Z","iopub.status.idle":"2024-12-14T06:44:26.521336Z","shell.execute_reply.started":"2024-12-14T06:44:26.516997Z","shell.execute_reply":"2024-12-14T06:44:26.520548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def crop_center_square(frame):\n    y,x = frame.shape[0:2]\n    min_dim = min(y, x)\n    start_x = (x // 2) - (min_dim // 2)\n    start_y = (y // 2) - (min_dim // 2)\n    return frame[start_y :start_y + min_dim, start_x : start_x + min_dim]\n\ndef load_video(path, max_frames=0, resize=(img_size, img_size)):\n    cap = cv2.VideoCapture(path)\n    frames = []\n    try:\n        while 1:\n            ret, frame = cap.read()\n            if not ret:\n                break\n            frame = crop_center_square(frame)\n            frame = cv2.resize(frame, resize)\n            frame = frame[:, :, [2, 1, 0]]\n            frames.append(frame)\n            \n            if len(frames) == max_frames:\n                break\n    finally:\n        cap.release()\n    return np.array(frames)\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:26.522778Z","iopub.execute_input":"2024-12-14T06:44:26.523068Z","iopub.status.idle":"2024-12-14T06:44:26.536124Z","shell.execute_reply.started":"2024-12-14T06:44:26.523017Z","shell.execute_reply":"2024-12-14T06:44:26.534937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pretrain_feature_extractor():\n    feature_extractor = keras.applications.InceptionV3(\n    weights = \"imagenet\",\n    include_top=False,\n    pooling=\"avg\",\n    input_shape = (img_size,img_size,3)\n    )\n    preprocess_input = keras.applications.inception_v3.preprocess_input\n    \n    inputs = keras.Input((img_size,img_size,3))\n    preprocessed = preprocess_input(inputs)\n    \n    outputs = feature_extractor(preprocessed)\n    return keras.Model(inputs, outputs, name=\"feature_extractor\")\n\nfeature_extractor = pretrain_feature_extractor()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:26.537972Z","iopub.execute_input":"2024-12-14T06:44:26.538485Z","iopub.status.idle":"2024-12-14T06:44:29.353165Z","shell.execute_reply.started":"2024-12-14T06:44:26.538286Z","shell.execute_reply":"2024-12-14T06:44:29.352445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_all_videos(df, root_dir):\n    num_samples = len(df)\n    video_paths = list(df.index)\n    labels = df[\"label\"].values\n    labels = np.array(labels=='FAKE').astype(np.int)\n    \n    frame_masks = np.zeros(shape=(num_samples, max_seq_length), dtype=\"bool\")\n    frame_features = np.zeros(\n        shape=(num_samples, max_seq_length, num_features), dtype=\"float32\" \n    )\n    \n    for idx, path in enumerate(video_paths):\n        frames = load_video(os.path.join(root_dir, path))\n        frames = frames[None, ...]\n        \n        temp_frame_mask = np.zeros(shape=(1, max_seq_length,), dtype=\"bool\")\n        temp_frame_features = np.zeros(shape=(1, max_seq_length, num_features), dtype=\"float32\")\n        \n        for i, batch in enumerate(frames):\n            video_length = batch.shape[0] \n            length = min(max_seq_length, video_length) #if length is over 20s ,only cut 20s\n            for j in range(length):\n                temp_frame_features[i, j, :] =feature_extractor.predict(batch[None, j, :])\n            temp_frame_mask[i, :length] =1 # 1 = not masked, 0 = masked ->give 1 when there are images ,otherwise 0 for padding\n        \n        frame_features[idx,] =temp_frame_features.squeeze() #squeeze array for training\n        frame_masks[idx,] =temp_frame_mask.squeeze()\n    \n    return (frame_features, frame_masks), labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:29.354180Z","iopub.execute_input":"2024-12-14T06:44:29.354434Z","iopub.status.idle":"2024-12-14T06:44:29.363352Z","shell.execute_reply.started":"2024-12-14T06:44:29.354392Z","shell.execute_reply":"2024-12-14T06:44:29.362617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set , Test_set = train_test_split(train_sample_metadata, test_size=0.2,random_state=42,\n                                       stratify=train_sample_metadata['label'])\nprint(Train_set.shape, Test_set.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:29.364593Z","iopub.execute_input":"2024-12-14T06:44:29.364830Z","iopub.status.idle":"2024-12-14T06:44:29.380647Z","shell.execute_reply.started":"2024-12-14T06:44:29.364783Z","shell.execute_reply":"2024-12-14T06:44:29.379934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, train_labels = prepare_all_videos(Train_set, \"train\")\ntest_data, test_labels = prepare_all_videos(Test_set, \"test\")\n\nprint(f\"Frame features in train set:{train_data[0].shape}\")\nprint(f\"Frame masks in train set:{train_data[1].shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:29.381741Z","iopub.execute_input":"2024-12-14T06:44:29.382015Z","iopub.status.idle":"2024-12-14T06:44:29.520578Z","shell.execute_reply.started":"2024-12-14T06:44:29.381963Z","shell.execute_reply":"2024-12-14T06:44:29.519722Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Training(RNN)**","metadata":{}},{"cell_type":"code","source":"\nframe_features_input = keras.Input((max_seq_length, num_features))\nmask_input = keras.Input((max_seq_length,),dtype=\"bool\")\n\nx = keras.layers.GRU(16, return_sequences=True)(frame_features_input, mask = mask_input)\nx = keras.layers.GRU(8)(x)\nx = keras.layers.Dropout(0.4)(x)\nx = keras.layers.Dense(8, activation=\"relu\")(x)\noutput = keras.layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = keras.Model([frame_features_input, mask_input], output)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\", keras.metrics.Precision()])\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:29.521922Z","iopub.execute_input":"2024-12-14T06:44:29.522252Z","iopub.status.idle":"2024-12-14T06:44:30.880534Z","shell.execute_reply.started":"2024-12-14T06:44:29.522191Z","shell.execute_reply":"2024-12-14T06:44:30.879725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint = keras.callbacks.ModelCheckpoint('./', save_weights_only=True, save_best_only=True)\nhistory = model.fit(\n        [train_data[0], train_data[1]],\n        train_labels,\n        validation_data=([test_data[0], test_data[1]], test_labels),\n        callbacks=[checkpoint],\n        epochs=epochs,\n        batch_size=64\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:30.881902Z","iopub.execute_input":"2024-12-14T06:44:30.882235Z","iopub.status.idle":"2024-12-14T06:44:40.274556Z","shell.execute_reply.started":"2024-12-14T06:44:30.882156Z","shell.execute_reply":"2024-12-14T06:44:40.273817Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Inference**","metadata":{}},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:40.275865Z","iopub.execute_input":"2024-12-14T06:44:40.276063Z","iopub.status.idle":"2024-12-14T06:44:40.282544Z","shell.execute_reply.started":"2024-12-14T06:44:40.276029Z","shell.execute_reply":"2024-12-14T06:44:40.281792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_single_video(frames):\n    frames = frames[None, ...]\n    frame_mask = np.zeros(shape=(1, max_seq_length,), dtype=\"bool\")\n    frame_features = np.zeros(shape=(1, max_seq_length, num_features), dtype=\"float32\")\n\n    for i, batch in enumerate(frames):\n        video_length = batch.shape[0]\n        length = min(max_seq_length, video_length)\n        for j in range(length):\n            frame_features[i, j, :] = feature_extractor.predict(batch[None, j, :])\n        frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n    return frame_features, frame_mask\n\ndef sequence_prediction(path):\n    frames = load_video(os.path.join(DATA_FOLDER, TEST_FOLDER,path))\n    frame_features, frame_mask = prepare_single_video(frames)\n    return model.predict([frame_features, frame_mask])[0]\n    \n# This utility is for visualization.\n# Referenced from:\n# https://www.tensorflow.org/hub/tutorials/action_recognition_with_tf_hub\ndef to_gif(images):\n    converted_images = images.astype(np.uint8)\n    imageio.mimsave(\"animation.gif\", converted_images, fps=10)\n    return embed.embed_file(\"animation.gif\")\n\n\ntest_video = np.random.choice(test_videos[\"video\"].values.tolist())\nprint(f\"Test video path: {test_video}\")\n\nif(sequence_prediction(test_video)>=0.5):\n    print(f'The predicted class of the video is FAKE')\nelse:\n    print(f'The predicted class of the video is REAL')\n\nplay_video(test_video,TEST_FOLDER)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:40.284308Z","iopub.execute_input":"2024-12-14T06:44:40.284760Z","iopub.status.idle":"2024-12-14T06:44:47.347446Z","shell.execute_reply.started":"2024-12-14T06:44:40.284578Z","shell.execute_reply":"2024-12-14T06:44:47.345792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**ĐÔ NAN TRĂM**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the root directory for the dataset\nroot_dir = \"/kaggle/input/deep-fake-detection-dfd-entire-original-dataset\"\n\n# Prepare data to write\ndata = []\n\n# Traverse the dataset directory\nfor label, folder in [(\"FAKE\", \"DFD_manipulated_sequences/DFD_manipulated_sequences\"), (\"REAL\", \"DFD_original sequences\")]:\n    label_dir = os.path.join(root_dir, folder)\n    if os.path.exists(label_dir):\n        for filename in os.listdir(label_dir):\n            # Only process files (exclude subdirectories, if any)\n            if os.path.isfile(os.path.join(label_dir, filename)):\n                data.append([filename, label, \"train\"])\n\n# Create a DataFrame\ncolumns = [\"filename\", \"label\", \"split\"]\nTest_set = pd.DataFrame(data, columns=columns)\nTest_set.set_index(\"filename\", inplace=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:47.349549Z","iopub.execute_input":"2024-12-14T06:44:47.350079Z","iopub.status.idle":"2024-12-14T06:44:55.330875Z","shell.execute_reply.started":"2024-12-14T06:44:47.349998Z","shell.execute_reply":"2024-12-14T06:44:55.330230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_FOLDER_2 = '/kaggle/input/deep-fake-detection-dfd-entire-original-dataset'\nfeature_extractor = pretrain_feature_extractor()\n# Read the file, assuming columns are separated by spaces\ntest_data, test_labels = prepare_all_videos(Test_set, \"deep-fake-detection-dfd-entire-original-dataset\")\n\nprint(f\"Frame features in train set:{train_data[0].shape}\")\nprint(f\"Frame masks in train set:{train_data[1].shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:55.332052Z","iopub.execute_input":"2024-12-14T06:44:55.332336Z","iopub.status.idle":"2024-12-14T06:44:59.130022Z","shell.execute_reply.started":"2024-12-14T06:44:55.332274Z","shell.execute_reply":"2024-12-14T06:44:59.129212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# checkpoint = keras.callbacks.ModelCheckpoint('./', save_weights_only=True, save_best_only=True)\nhistory = model.fit(\n        [train_data[0], train_data[1]],\n        train_labels,\n        validation_data=([test_data[0], test_data[1]], test_labels),\n        # callbacks=[checkpoint],\n        epochs=epochs,\n        batch_size=64\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:44:59.131677Z","iopub.execute_input":"2024-12-14T06:44:59.132028Z","iopub.status.idle":"2024-12-14T06:45:11.560729Z","shell.execute_reply.started":"2024-12-14T06:44:59.131964Z","shell.execute_reply":"2024-12-14T06:45:11.559918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '/kaggle/input/deep-fake-detection-dfd-entire-original-dataset'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'DFD_original sequences'\ntest_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])\nprint(test_videos)\ndef prepare_single_video(frames):\n    frames = frames[None, ...]\n    frame_mask = np.zeros(shape=(1, max_seq_length,), dtype=\"bool\")\n    frame_features = np.zeros(shape=(1, max_seq_length, num_features), dtype=\"float32\")\n\n    for i, batch in enumerate(frames):\n        video_length = batch.shape[0]\n        length = min(max_seq_length, video_length)\n        for j in range(length):\n            frame_features[i, j, :] = feature_extractor.predict(batch[None, j, :])\n        frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n    return frame_features, frame_mask\n\ndef sequence_prediction(path):\n    frames = load_video(os.path.join(DATA_FOLDER, TEST_FOLDER,path))\n    frame_features, frame_mask = prepare_single_video(frames)\n    return model.predict([frame_features, frame_mask])[0]\n    \n# This utility is for visualization.\n# Referenced from:\n# https://www.tensorflow.org/hub/tutorials/action_recognition_with_tf_hub\ndef to_gif(images):\n    converted_images = images.astype(np.uint8)\n    imageio.mimsave(\"animation.gif\", converted_images, fps=10)\n    return embed.embed_file(\"animation.gif\")\n\n\ntest_video = np.random.choice(test_videos[\"video\"].values.tolist())\nprint(f\"Test video path: {test_video}\")\n\nif(sequence_prediction(test_video)>=0.5):\n    print(f'The predicted class of the video is FAKE')\nelse:\n    print(f'The predicted class of the video is REAL')\n\nplay_video(test_video,TEST_FOLDER)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:45:11.562217Z","iopub.execute_input":"2024-12-14T06:45:11.562544Z","iopub.status.idle":"2024-12-14T06:45:18.091147Z","shell.execute_reply.started":"2024-12-14T06:45:11.562488Z","shell.execute_reply":"2024-12-14T06:45:18.088404Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Homemade","metadata":{}},{"cell_type":"code","source":"import os\nimport csv\n\n# Define the root directory for the dataset\nroot_dir = \"/kaggle/input/homemade/deepfake data homemade\"\n\n# Define the output CSV file\noutput_csv = \"metadata.csv\"\n\n# Prepare the CSV headers\nheaders = [\"filename\", \"label\", \"split\"]\n\n# Prepare data to write\ndata = []\n\n# Traverse the dataset directory\n\nfor label, folder in [(\"FAKE\", \"fake\"), (\"REAL\", \"real\")]:\n    label_dir = os.path.join(root_dir, folder)\n    if os.path.exists(label_dir):\n        for filename in os.listdir(label_dir):\n            # Only process files (exclude subdirectories, if any)\n            if os.path.isfile(os.path.join(label_dir, filename)):\n                data.append([filename, label.upper(), \"test\"])\n                \ncolumns = [\"filename\", \"label\", \"split\"]\n\nTest_set = pd.DataFrame(data, columns=columns)\nTest_set.set_index(\"filename\", inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:45:18.093598Z","iopub.execute_input":"2024-12-14T06:45:18.094083Z","iopub.status.idle":"2024-12-14T06:45:18.130826Z","shell.execute_reply.started":"2024-12-14T06:45:18.094000Z","shell.execute_reply":"2024-12-14T06:45:18.130003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Test_set","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:45:18.132008Z","iopub.execute_input":"2024-12-14T06:45:18.132281Z","iopub.status.idle":"2024-12-14T06:45:18.141273Z","shell.execute_reply.started":"2024-12-14T06:45:18.132234Z","shell.execute_reply":"2024-12-14T06:45:18.140451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# checkpoint = keras.callbacks.ModelCheckpoint('./', save_weights_only=True, save_best_only=True)\nhistory = model.fit(\n        [train_data[0], train_data[1]],\n        train_labels,\n        validation_data=([test_data[0], test_data[1]], test_labels),\n        # callbacks=[checkpoint],\n        epochs=epochs,\n        batch_size=64\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:45:18.143152Z","iopub.execute_input":"2024-12-14T06:45:18.143529Z","iopub.status.idle":"2024-12-14T06:45:30.876676Z","shell.execute_reply.started":"2024-12-14T06:45:18.143460Z","shell.execute_reply":"2024-12-14T06:45:30.875723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '/kaggle/input/homemade/deepfake data homemade'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'real'\ntest_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])\nprint(test_videos)\ndef prepare_single_video(frames):\n    frames = frames[None, ...]\n    frame_mask = np.zeros(shape=(1, max_seq_length,), dtype=\"bool\")\n    frame_features = np.zeros(shape=(1, max_seq_length, num_features), dtype=\"float32\")\n\n    for i, batch in enumerate(frames):\n        video_length = batch.shape[0]\n        length = min(max_seq_length, video_length)\n        for j in range(length):\n            frame_features[i, j, :] = feature_extractor.predict(batch[None, j, :])\n        frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n    return frame_features, frame_mask\n\ndef sequence_prediction(path):\n    frames = load_video(os.path.join(DATA_FOLDER, TEST_FOLDER,path))\n    frame_features, frame_mask = prepare_single_video(frames)\n    return model.predict([frame_features, frame_mask])[0]\n    \n# This utility is for visualization.\n# Referenced from:\n# https://www.tensorflow.org/hub/tutorials/action_recognition_with_tf_hub\ndef to_gif(images):\n    converted_images = images.astype(np.uint8)\n    imageio.mimsave(\"animation.gif\", converted_images, fps=10)\n    return embed.embed_file(\"animation.gif\")\n\n\ntest_video = np.random.choice(test_videos[\"video\"].values.tolist())\nprint(f\"Test video path: {test_video}\")\n\nif(sequence_prediction(test_video)>=0.5):\n    print(f'The predicted class of the video is FAKE')\nelse:\n    print(f'The predicted class of the video is REAL')\n\nplay_video(test_video,TEST_FOLDER)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:45:30.877967Z","iopub.execute_input":"2024-12-14T06:45:30.878233Z","iopub.status.idle":"2024-12-14T06:45:47.840930Z","shell.execute_reply.started":"2024-12-14T06:45:30.878182Z","shell.execute_reply":"2024-12-14T06:45:47.839207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}