{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":924245,"sourceType":"datasetVersion","datasetId":464091}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -U --upgrade tensorflow","metadata":{"_uuid":"40b2707f-0ecf-44f0-8af1-e982fea1b62f","_cell_guid":"1f8d3de6-788d-43fd-99bd-e494b49f20c6","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**this project is still under process**","metadata":{}},{"cell_type":"code","source":"# from tensorflow_docs.vis import embed\nfrom tensorflow import keras\n#from imutils import paths\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"_uuid":"636f9903-badf-43dc-a78f-92e6bd064e56","_cell_guid":"87b2fdde-f117-46db-a8c7-7b672ea6be42","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:35:56.631520Z","iopub.execute_input":"2024-08-15T16:35:56.631912Z","iopub.status.idle":"2024-08-15T16:35:59.827578Z","shell.execute_reply.started":"2024-08-15T16:35:56.631854Z","shell.execute_reply":"2024-08-15T16:35:59.826739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = '/kaggle/input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\nTEST_FOLDER = '/kaggle/input/deepfake-detection-challenge/test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"_uuid":"ddb9afb9-cb60-46ef-b941-c4adea7f449e","_cell_guid":"67fd3710-db1d-4d34-b517-d69122120113","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:35:59.830189Z","iopub.execute_input":"2024-08-15T16:35:59.830599Z","iopub.status.idle":"2024-08-15T16:35:59.838760Z","shell.execute_reply.started":"2024-08-15T16:35:59.830526Z","shell.execute_reply":"2024-08-15T16:35:59.837770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"_uuid":"a53bdc2d-ae68-4c3a-9ca7-604d024c3f22","_cell_guid":"c4193697-bce7-4e31-a047-009a38c7b8d8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:35:59.840476Z","iopub.execute_input":"2024-08-15T16:35:59.840843Z","iopub.status.idle":"2024-08-15T16:36:00.123933Z","shell.execute_reply.started":"2024-08-15T16:35:59.840780Z","shell.execute_reply":"2024-08-15T16:36:00.122885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()","metadata":{"_uuid":"c8e1b13e-fd21-4edd-8972-5c536253cf49","_cell_guid":"f47433a8-2a83-4edb-a3b7-c62a8b661e8d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:00.125749Z","iopub.execute_input":"2024-08-15T16:36:00.126085Z","iopub.status.idle":"2024-08-15T16:36:00.468165Z","shell.execute_reply.started":"2024-08-15T16:36:00.126022Z","shell.execute_reply":"2024-08-15T16:36:00.466292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata.shape","metadata":{"_uuid":"b029ff6a-d177-4a95-9f08-a66f09a29c71","_cell_guid":"63d0caac-fc34-4541-b3e8-25ba2369e805","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:00.472330Z","iopub.execute_input":"2024-08-15T16:36:00.473401Z","iopub.status.idle":"2024-08-15T16:36:00.483094Z","shell.execute_reply.started":"2024-08-15T16:36:00.473295Z","shell.execute_reply":"2024-08-15T16:36:00.481507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(10).index)\nfake_train_sample_video","metadata":{"_uuid":"332315d8-8728-4ac3-b4d1-68e15a9b1314","_cell_guid":"2597b901-47d5-4101-ac6a-e0709920ab5b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:00.486409Z","iopub.execute_input":"2024-08-15T16:36:00.487492Z","iopub.status.idle":"2024-08-15T16:36:00.503611Z","shell.execute_reply.started":"2024-08-15T16:36:00.487393Z","shell.execute_reply":"2024-08-15T16:36:00.502005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    \n    capture_image = cv2.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"_uuid":"bd1742a4-fd30-4d15-90fb-f5f6584578bf","_cell_guid":"2689d44c-9408-4c01-a6e7-fe471e88d7fa","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:02.091362Z","iopub.execute_input":"2024-08-15T16:36:02.091738Z","iopub.status.idle":"2024-08-15T16:36:02.098760Z","shell.execute_reply.started":"2024-08-15T16:36:02.091659Z","shell.execute_reply":"2024-08-15T16:36:02.097615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"_uuid":"5b55874a-d919-4a4d-be0a-9ce18f64f816","_cell_guid":"6ee33657-1611-49ae-bdc7-c409c82054fa","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:02.954386Z","iopub.execute_input":"2024-08-15T16:36:02.954810Z","iopub.status.idle":"2024-08-15T16:36:10.104972Z","shell.execute_reply.started":"2024-08-15T16:36:02.954749Z","shell.execute_reply":"2024-08-15T16:36:10.103872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(3).index)\nreal_train_sample_video","metadata":{"_uuid":"caee50b0-2b46-41ed-9014-fc711dbcf0ac","_cell_guid":"f6fc6ddd-a52f-4d60-8dd1-14e1fe2d119c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:10.107237Z","iopub.execute_input":"2024-08-15T16:36:10.107870Z","iopub.status.idle":"2024-08-15T16:36:10.118600Z","shell.execute_reply.started":"2024-08-15T16:36:10.107801Z","shell.execute_reply":"2024-08-15T16:36:10.117574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"_uuid":"99d14696-3339-4a8b-911d-7bcfa07e3376","_cell_guid":"3fc4ad65-9092-4303-857d-329830dab4e3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:10.120425Z","iopub.execute_input":"2024-08-15T16:36:10.121085Z","iopub.status.idle":"2024-08-15T16:36:12.273010Z","shell.execute_reply.started":"2024-08-15T16:36:10.121012Z","shell.execute_reply":"2024-08-15T16:36:12.272024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata['original'].value_counts()[0:5]","metadata":{"_uuid":"630679cf-4f49-4e32-97b7-fe961748f10f","_cell_guid":"33b6e923-f0e5-43aa-8f5f-51982c059de7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:12.274920Z","iopub.execute_input":"2024-08-15T16:36:12.275301Z","iopub.status.idle":"2024-08-15T16:36:12.287554Z","shell.execute_reply.started":"2024-08-15T16:36:12.275242Z","shell.execute_reply":"2024-08-15T16:36:12.286612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    \n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    # we only show images extracted from the first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n        video_path = os.path.join(DATA_FOLDER, video_folder,video_file)\n        capture_image = cv2.VideoCapture(video_path) \n        ret, frame = capture_image.read()\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        ax[i//3, i%3].imshow(frame)\n        ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n        ax[i//3, i%3].axis('on')","metadata":{"_uuid":"35116976-5862-49cb-b42a-9ca754dabf08","_cell_guid":"5d26d2a5-2971-4d9e-b9e7-2e7723955353","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:12.290307Z","iopub.execute_input":"2024-08-15T16:36:12.290658Z","iopub.status.idle":"2024-08-15T16:36:12.302516Z","shell.execute_reply.started":"2024-08-15T16:36:12.290595Z","shell.execute_reply":"2024-08-15T16:36:12.301364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='atvmxvwyns.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"_uuid":"41df3fff-e47e-414b-a256-49ee29a9eed6","_cell_guid":"f3b28bf6-5245-4869-90ff-b7a9725f8c94","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:12.304565Z","iopub.execute_input":"2024-08-15T16:36:12.304951Z","iopub.status.idle":"2024-08-15T16:36:14.411826Z","shell.execute_reply.started":"2024-08-15T16:36:12.304885Z","shell.execute_reply":"2024-08-15T16:36:14.410742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"_uuid":"a453365f-da8a-4bbd-9fa7-e30bb9c15d3e","_cell_guid":"2aff5cd4-a624-4a19-b06a-711ae95533a6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:14.413975Z","iopub.execute_input":"2024-08-15T16:36:14.414270Z","iopub.status.idle":"2024-08-15T16:36:14.420653Z","shell.execute_reply.started":"2024-08-15T16:36:14.414222Z","shell.execute_reply":"2024-08-15T16:36:14.419751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos.head()","metadata":{"_uuid":"84a663d1-1bc4-4d5f-89a8-cf977735ae98","_cell_guid":"a1f05035-1e8b-4450-9519-4891213445de","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:15.018285Z","iopub.execute_input":"2024-08-15T16:36:15.018655Z","iopub.status.idle":"2024-08-15T16:36:15.028928Z","shell.execute_reply.started":"2024-08-15T16:36:15.018602Z","shell.execute_reply":"2024-08-15T16:36:15.028030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_from_video(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[2].video))","metadata":{"_uuid":"902773be-ad4c-4e50-bb8e-2a92dfdc18d8","_cell_guid":"4881d631-3be2-4248-a8a3-323ac5bb9ae8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:18.611367Z","iopub.execute_input":"2024-08-15T16:36:18.611724Z","iopub.status.idle":"2024-08-15T16:36:19.503630Z","shell.execute_reply.started":"2024-08-15T16:36:18.611658Z","shell.execute_reply":"2024-08-15T16:36:19.502640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_videos = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].index)","metadata":{"_uuid":"73471c92-47c4-4422-89dd-22636ea8bfc6","_cell_guid":"5e4be99f-dc6d-45f4-9cad-040e2a28a2ab","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:20.171541Z","iopub.execute_input":"2024-08-15T16:36:20.171956Z","iopub.status.idle":"2024-08-15T16:36:20.178166Z","shell.execute_reply.started":"2024-08-15T16:36:20.171884Z","shell.execute_reply":"2024-08-15T16:36:20.177265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\n\ndef play_video(video_file, subset=TRAIN_SAMPLE_FOLDER):\n    \n    video_url = open(os.path.join(DATA_FOLDER, subset,video_file),'rb').read()\n    data_url = \"data:video/mp4;base64,\" + b64encode(video_url).decode()\n    return HTML(\"\"\"<video width=500 controls><source src=\"%s\" type=\"video/mp4\"></video>\"\"\" % data_url)\n\nplay_video(fake_videos[10])","metadata":{"_uuid":"cdf5856d-fa58-41df-a3b5-1a8d0ae779ba","_cell_guid":"9aa950f1-afe6-42ea-9763-6d6619de54c1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:23.611354Z","iopub.execute_input":"2024-08-15T16:36:23.611730Z","iopub.status.idle":"2024-08-15T16:36:23.782493Z","shell.execute_reply.started":"2024-08-15T16:36:23.611656Z","shell.execute_reply":"2024-08-15T16:36:23.780942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 64\nEPOCHS = 10\n\nMAX_SEQ_LENGTH = 20\nNUM_FEATURES = 2048","metadata":{"_uuid":"8f525cda-bc46-4072-bcc1-63fdac3cfbe7","_cell_guid":"7d5655f9-dfdc-4c45-be4e-550813a89efd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:36:59.051288Z","iopub.execute_input":"2024-08-15T16:36:59.051910Z","iopub.status.idle":"2024-08-15T16:36:59.056903Z","shell.execute_reply.started":"2024-08-15T16:36:59.051646Z","shell.execute_reply":"2024-08-15T16:36:59.055973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_center_square(frame):\n    y, x = frame.shape[0:2]\n    min_dim = min(y, x)\n    start_x = (x // 2) - (min_dim // 2)\n    start_y = (y // 2) - (min_dim // 2)\n    return frame[start_y : start_y + min_dim, start_x : start_x + min_dim]\n\n\ndef load_video(path, max_frames=0, resize=(IMG_SIZE, IMG_SIZE)):\n    cap = cv2.VideoCapture(path)\n    frames = []\n    try:\n        while True:\n            ret, frame = cap.read()\n            if not ret:\n                break\n            frame = crop_center_square(frame)\n            frame = cv2.resize(frame, resize)\n            frame = frame[:, :, [2, 1, 0]]\n            frames.append(frame)\n\n            if len(frames) == max_frames:\n                break\n    finally:\n        cap.release()\n    return np.array(frames)","metadata":{"_uuid":"aeb573b8-8e1a-4a3c-9b09-6213ce6d49e2","_cell_guid":"b41ab383-9c56-429f-bdf8-960b1a3c99ba","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:40:15.863579Z","iopub.execute_input":"2024-08-15T16:40:15.863988Z","iopub.status.idle":"2024-08-15T16:40:15.877136Z","shell.execute_reply.started":"2024-08-15T16:40:15.863930Z","shell.execute_reply":"2024-08-15T16:40:15.876163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_feature_extractor():\n    feature_extractor = keras.applications.InceptionV3(\n        weights=\"imagenet\",\n        include_top=False,\n        pooling=\"avg\",\n        input_shape=(IMG_SIZE, IMG_SIZE, 3),\n    )\n    preprocess_input = keras.applications.inception_v3.preprocess_input\n\n    inputs = keras.Input((IMG_SIZE, IMG_SIZE, 3))\n    preprocessed = preprocess_input(inputs)\n\n    outputs = feature_extractor(preprocessed)\n    return keras.Model(inputs, outputs, name=\"feature_extractor\")\n\n\nfeature_extractor = build_feature_extractor()","metadata":{"_uuid":"99f51d96-8b4f-4120-9cd2-b4fb67ed8da6","_cell_guid":"54ce3e12-f967-4244-a546-1e39fbcefdfb","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:40:19.173783Z","iopub.execute_input":"2024-08-15T16:40:19.174142Z","iopub.status.idle":"2024-08-15T16:40:23.624565Z","shell.execute_reply.started":"2024-08-15T16:40:19.174091Z","shell.execute_reply":"2024-08-15T16:40:23.623796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_all_videos(df, root_dir):\n    num_samples = len(df)\n    video_paths = list(df.index)\n    labels = df[\"label\"].values\n    labels = np.array(labels=='FAKE').astype(int)\n\n    frame_masks = np.zeros(shape=(num_samples, MAX_SEQ_LENGTH), dtype=\"bool\")\n    frame_features = np.zeros(\n        shape=(num_samples, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\"\n    )\n\n    # For each video.\n    for idx, path in enumerate(video_paths):\n        # Gather all its frames and add a batch dimension.\n        frames = load_video(os.path.join(root_dir, path))\n        frames = frames[None, ...]\n\n        # Initialize placeholders to store the masks and features of the current video.\n        temp_frame_mask = np.zeros(shape=(1, MAX_SEQ_LENGTH,), dtype=\"bool\")\n        temp_frame_features = np.zeros(\n            shape=(1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\"\n        )\n\n        # Extract features from the frames of the current video.\n        for i, batch in enumerate(frames):\n            video_length = batch.shape[0]\n            length = min(MAX_SEQ_LENGTH, video_length)\n            for j in range(length):\n                temp_frame_features[i, j, :] = feature_extractor.predict(\n                    batch[None, j, :]\n                )\n            temp_frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n        frame_features[idx,] = temp_frame_features.squeeze()\n        frame_masks[idx,] = temp_frame_mask.squeeze()\n\n    return (frame_features, frame_masks), labels","metadata":{"_uuid":"b95bdb68-8466-4301-9dfc-d31ffb2b2020","_cell_guid":"1d9b620e-364c-4fc6-8da6-9a6bdaf83d8d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:40:23.626980Z","iopub.execute_input":"2024-08-15T16:40:23.627396Z","iopub.status.idle":"2024-08-15T16:40:23.643810Z","shell.execute_reply.started":"2024-08-15T16:40:23.627324Z","shell.execute_reply":"2024-08-15T16:40:23.642613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot training & validation accuracy values\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\nplt.show()\n\n# Plot training & validation loss values\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\nplt.show()","metadata":{"_uuid":"094bde13-b61f-44ec-adcf-7f4af3d9e0eb","_cell_guid":"9aa9a00f-201a-40b2-becf-712541902871","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:39:59.901433Z","iopub.execute_input":"2024-08-15T16:39:59.901838Z","iopub.status.idle":"2024-08-15T16:40:00.426465Z","shell.execute_reply.started":"2024-08-15T16:39:59.901778Z","shell.execute_reply":"2024-08-15T16:40:00.425330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_single_video(frames):\n    frames = frames[None, ...]\n    frame_mask = np.zeros(shape=(1, MAX_SEQ_LENGTH,), dtype=\"bool\")\n    frame_features = np.zeros(shape=(1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    for i, batch in enumerate(frames):\n        video_length = batch.shape[0]\n        length = min(MAX_SEQ_LENGTH, video_length)\n        for j in range(length):\n            frame_features[i, j, :] = feature_extractor.predict(batch[None, j, :])\n        frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n    return frame_features, frame_mask\n\ndef sequence_prediction(path):\n    frames = load_video(os.path.join(DATA_FOLDER, TEST_FOLDER,path))\n    frame_features, frame_mask = prepare_single_video(frames)\n    return model.predict([frame_features, frame_mask])[0]\n    \n# This utility is for visualization.\n# Referenced from:\n# https://www.tensorflow.org/hub/tutorials/action_recognition_with_tf_hub\ndef to_gif(images):\n    converted_images = images.astype(np.uint8)\n    imageio.mimsave(\"animation.gif\", converted_images, fps=30)\n    return embed.embed_file(\"animation.gif\")","metadata":{"_uuid":"5e26f0cb-6926-4bbd-b4d1-93225829b081","_cell_guid":"fa7bff6b-8802-43cd-8cf9-c0531fd757f5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:39:47.669218Z","iopub.execute_input":"2024-08-15T16:39:47.669583Z","iopub.status.idle":"2024-08-15T16:39:47.682883Z","shell.execute_reply.started":"2024-08-15T16:39:47.669531Z","shell.execute_reply":"2024-08-15T16:39:47.681786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_video = \"/kaggle/input/deepfake-detection-challenge/train_sample_videos/aapnvogymq.mp4\"\nprint(f\"Test video path: {test_video}\")\n\nif(sequence_prediction(test_video)<=0.6):\n    print(f'The predicted class of the video is FAKE')\nelse:\n    print(f'The predicted class of the video is REAL')\nprint(sequence_prediction(test_video))\nplay_video(test_video,TEST_FOLDER)","metadata":{"_uuid":"09cb3e78-ff2d-49e4-831e-b974660704bf","_cell_guid":"e0d85cac-3095-46f4-9e49-6ff784e3c3e7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:54:40.917461Z","iopub.execute_input":"2024-08-15T16:54:40.917895Z","iopub.status.idle":"2024-08-15T16:54:49.649216Z","shell.execute_reply.started":"2024-08-15T16:54:40.917822Z","shell.execute_reply":"2024-08-15T16:54:49.646798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_video = \"/kaggle/input/deepfake-detection-challenge/train_sample_videos/acifjvzvpm.mp4\"\nprint(f\"Test video path: {test_video}\")\n\nif(sequence_prediction(test_video)<=0.6):\n    print(f'The predicted class of the video is FAKE')\nelse:\n    print(f'The predicted class of the video is REAL')\n    \nprint(sequence_prediction(test_video))\nplay_video(test_video,TEST_FOLDER)","metadata":{"_uuid":"3f32f983-9d94-4fc7-afab-08f1cd7e6c8d","_cell_guid":"daa6c50b-0dea-46f9-84b4-1e8c5a1910b1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:55:19.591641Z","iopub.execute_input":"2024-08-15T16:55:19.592079Z","iopub.status.idle":"2024-08-15T16:55:28.659943Z","shell.execute_reply.started":"2024-08-15T16:55:19.592007Z","shell.execute_reply":"2024-08-15T16:55:28.657302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom tensorflow import keras\nfrom tensorflow.keras.callbacks import EarlyStopping, LearningRateScheduler\nimport numpy as np\n\ndef build_model():\n    from keras.layers import Bidirectional, BatchNormalization\n\n    frame_features_input = keras.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\n    mask_input = keras.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n\n    # Increase model complexity\n    x = Bidirectional(keras.layers.GRU(128, return_sequences=True))(frame_features_input, mask=mask_input)\n    x = BatchNormalization()(x)\n    x = Bidirectional(keras.layers.GRU(64))(x)\n    x = keras.layers.Dropout(0.5)(x)  # Increased dropout rate\n\n    # Add more dense layers\n    x = keras.layers.Dense(64, activation=\"relu\")(x)\n    x = BatchNormalization()(x)\n    x = keras.layers.Dense(32, activation=\"relu\")(x)\n\n    # Output layer\n    output = keras.layers.Dense(1, activation=\"sigmoid\")(x)\n\n    # Model definition\n    model = keras.Model([frame_features_input, mask_input], output)\n\n    # Compile the model with a modified optimizer\n    optimizer = keras.optimizers.Adam(learning_rate=0.001)\n    model.compile(loss=\"binary_crossentropy\", optimizer=optimizer, metrics=[\"accuracy\"])\n\n    # Model summary\n    return model\n\n\ndef scheduler(epoch, lr):\n    if epoch < 10:\n        return lr\n    else:\n        return lr * tf.math.exp(-0.1)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nall_features = np.random.randn(100, MAX_SEQ_LENGTH, NUM_FEATURES)  \nall_labels = np.random.randint(2, size=(100, 1))                  \n\nfold_no = 1\nfor train_index, test_index in kf.split(all_features):\n    train_features, test_features = all_features[train_index], all_features[test_index]\n    train_labels, test_labels = all_labels[train_index], all_labels[test_index]\n    \n    train_mask = np.ones((train_features.shape[0], MAX_SEQ_LENGTH), dtype=bool)\n    test_mask = np.ones((test_features.shape[0], MAX_SEQ_LENGTH), dtype=bool)\n    \n    model = build_model()\n\n    lr_scheduler = LearningRateScheduler(scheduler)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5)\n\n    print(f'Training for fold {fold_no}...')\n    history = model.fit(\n        [train_features, train_mask], \n        train_labels,\n        validation_data=([test_features, test_mask], test_labels),\n        epochs=30,\n        batch_size=10,\n        callbacks=[lr_scheduler, early_stopping]\n    )\n    \n    fold_no += 1","metadata":{"_uuid":"9bb213fa-05ac-4a67-88d0-fc78823f8773","_cell_guid":"b89793c5-c7cf-43fb-bf05-6c1d7e7e55ed","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:49:27.287251Z","iopub.execute_input":"2024-08-15T16:49:27.287668Z","iopub.status.idle":"2024-08-15T16:51:27.214873Z","shell.execute_reply.started":"2024-08-15T16:49:27.287596Z","shell.execute_reply":"2024-08-15T16:51:27.213969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"_uuid":"630f1fa3-48ac-4952-a0e9-02eae7e56197","_cell_guid":"4b5196d2-a8a9-4c51-8d3c-06de61e6823d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-08-15T16:45:47.093922Z","iopub.execute_input":"2024-08-15T16:45:47.094360Z","iopub.status.idle":"2024-08-15T16:45:47.106364Z","shell.execute_reply.started":"2024-08-15T16:45:47.094287Z","shell.execute_reply":"2024-08-15T16:45:47.105321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom tensorflow import keras\nfrom tensorflow.keras.callbacks import EarlyStopping, LearningRateScheduler\nimport numpy as np\n\ndef build_model():\n    frame_features_input = keras.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\n    mask_input = keras.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n    \n    x = keras.layers.GRU(64, return_sequences=True)(frame_features_input, mask=mask_input)\n    x = keras.layers.GRU(32)(x)\n    x = keras.layers.Dropout(0.2)(x)\n    x = keras.layers.Dense(32, activation=\"relu\")(x)\n    output = keras.layers.Dense(1, activation=\"sigmoid\")(x)\n    \n    model = keras.Model([frame_features_input, mask_input], output)\n    model.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\n    return model\n\ndef scheduler(epoch, lr):\n    if epoch < 10:\n        return lr\n    else:\n        return lr * tf.math.exp(-0.1)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nall_features = np.random.randn(100, MAX_SEQ_LENGTH, NUM_FEATURES)  \nall_labels = np.random.randint(2, size=(100, 1))                  \n\nfold_no = 1\nfor train_index, test_index in kf.split(all_features):\n    train_features, test_features = all_features[train_index], all_features[test_index]\n    train_labels, test_labels = all_labels[train_index], all_labels[test_index]\n    \n    train_mask = np.ones((train_features.shape[0], MAX_SEQ_LENGTH), dtype=bool)\n    test_mask = np.ones((test_features.shape[0], MAX_SEQ_LENGTH), dtype=bool)\n    \n    model = build_model()\n\n    lr_scheduler = LearningRateScheduler(scheduler)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=5)\n\n    print(f'Training for fold {fold_no}...')\n    history = model.fit(\n        [train_features, train_mask], \n        train_labels,\n        validation_data=([test_features, test_mask], test_labels),\n        epochs=30,\n        batch_size=10,\n        callbacks=[lr_scheduler, early_stopping]\n    )\n    \n    fold_no += 1","metadata":{"_uuid":"07503b2b-319a-4e2b-94c1-6fedc37119c9","_cell_guid":"21be8a40-694c-4a9f-a14b-c73ae4c37fe4","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"_uuid":"1b45902c-76e3-4fff-a44c-e75c219951d8","_cell_guid":"940d76ab-15cf-4c39-9eec-6bd2e0b8cc76","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"b00f677f-9c4a-4748-8f10-5cad68c41f03","_cell_guid":"80070522-fc6e-495a-9a81-3b524578422a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}