{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":924245,"sourceType":"datasetVersion","datasetId":464091}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Title","metadata":{"id":"Q6tasuafvT2O"}},{"cell_type":"markdown","source":"Deep Fake Image and Video Detection using CNN's and RNN's","metadata":{"id":"LvJBeHv3vfE_"}},{"cell_type":"markdown","source":"# Deep Fake Video Classification","metadata":{}},{"cell_type":"markdown","source":"## About the dataset","metadata":{}},{"cell_type":"markdown","source":"Files\n\n* train_sample_videos.zip - a ZIP file containing a sample set of training videos and a metadata.json with labels. the full set of training videos is available through the links provided above.\n* sample_submission.csv - a sample submission file in the correct format.\n* test_videos.zip - a zip file containing a small set of videos to be used as a public validation set. To understand the datasets available for this competition, review the Getting Started information.\n\nMetadata Columns\n\n* filename - the filename of the video\n* label - whether the video is REAL or FAKE\n* original - in the case that a train set video is FAKE, the original video is listed here\n* split - this is always equal to \"train\".","metadata":{}},{"cell_type":"markdown","source":"## Importing required libraries","metadata":{}},{"cell_type":"code","source":"!pip install -U --upgrade tensorflow","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:50:44.084717Z","iopub.execute_input":"2024-08-29T09:50:44.085094Z","iopub.status.idle":"2024-08-29T09:52:11.181696Z","shell.execute_reply.started":"2024-08-29T09:50:44.085015Z","shell.execute_reply":"2024-08-29T09:52:11.180859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from tensorflow_docs.vis import embed\nfrom tensorflow import keras\n#from imutils import paths\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:20.795342Z","iopub.execute_input":"2024-08-29T09:54:20.795743Z","iopub.status.idle":"2024-08-29T09:54:24.212869Z","shell.execute_reply.started":"2024-08-29T09:54:20.79567Z","shell.execute_reply":"2024-08-29T09:54:24.211958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Visualisation","metadata":{}},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:25.576669Z","iopub.execute_input":"2024-08-29T09:54:25.57698Z","iopub.status.idle":"2024-08-29T09:54:25.642742Z","shell.execute_reply.started":"2024-08-29T09:54:25.576938Z","shell.execute_reply":"2024-08-29T09:54:25.641978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:26.252604Z","iopub.execute_input":"2024-08-29T09:54:26.252942Z","iopub.status.idle":"2024-08-29T09:54:26.576634Z","shell.execute_reply.started":"2024-08-29T09:54:26.252898Z","shell.execute_reply":"2024-08-29T09:54:26.575779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:26.803771Z","iopub.execute_input":"2024-08-29T09:54:26.804107Z","iopub.status.idle":"2024-08-29T09:54:27.084873Z","shell.execute_reply.started":"2024-08-29T09:54:26.80405Z","shell.execute_reply":"2024-08-29T09:54:27.083789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata.shape","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:28.189196Z","iopub.execute_input":"2024-08-29T09:54:28.189521Z","iopub.status.idle":"2024-08-29T09:54:28.195214Z","shell.execute_reply.started":"2024-08-29T09:54:28.189476Z","shell.execute_reply":"2024-08-29T09:54:28.194438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualize now the data.\n\nWe select first a list of fake videos.","metadata":{}},{"cell_type":"markdown","source":"### Few fake videos","metadata":{}},{"cell_type":"code","source":"fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(3).index)\nfake_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:29.337842Z","iopub.execute_input":"2024-08-29T09:54:29.338146Z","iopub.status.idle":"2024-08-29T09:54:29.345981Z","shell.execute_reply.started":"2024-08-29T09:54:29.338103Z","shell.execute_reply":"2024-08-29T09:54:29.345204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_image = cv2.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:29.571586Z","iopub.execute_input":"2024-08-29T09:54:29.571896Z","iopub.status.idle":"2024-08-29T09:54:29.578588Z","shell.execute_reply.started":"2024-08-29T09:54:29.571853Z","shell.execute_reply":"2024-08-29T09:54:29.577705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:29.857821Z","iopub.execute_input":"2024-08-29T09:54:29.858138Z","iopub.status.idle":"2024-08-29T09:54:31.79811Z","shell.execute_reply.started":"2024-08-29T09:54:29.858095Z","shell.execute_reply":"2024-08-29T09:54:31.797275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try now the same for few of the images that are real.","metadata":{}},{"cell_type":"markdown","source":"### Few Real Videos","metadata":{}},{"cell_type":"code","source":"real_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(3).index)\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:31.799892Z","iopub.execute_input":"2024-08-29T09:54:31.800129Z","iopub.status.idle":"2024-08-29T09:54:31.80742Z","shell.execute_reply.started":"2024-08-29T09:54:31.800089Z","shell.execute_reply":"2024-08-29T09:54:31.806519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:31.808944Z","iopub.execute_input":"2024-08-29T09:54:31.809381Z","iopub.status.idle":"2024-08-29T09:54:33.64833Z","shell.execute_reply.started":"2024-08-29T09:54:31.809315Z","shell.execute_reply":"2024-08-29T09:54:33.647571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Videos with same original","metadata":{}},{"cell_type":"markdown","source":"Let's look now to set of samples with the same original.","metadata":{}},{"cell_type":"code","source":"train_sample_metadata['original'].value_counts()[0:5]","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:33.650006Z","iopub.execute_input":"2024-08-29T09:54:33.650284Z","iopub.status.idle":"2024-08-29T09:54:33.659175Z","shell.execute_reply.started":"2024-08-29T09:54:33.650236Z","shell.execute_reply":"2024-08-29T09:54:33.658358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We pick one of the originals with largest number of samples.\n\nWe also modify our visualization function to work with multiple images.","metadata":{}},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    # we only show images extracted from the first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n        video_path = os.path.join(DATA_FOLDER, video_folder,video_file)\n        capture_image = cv2.VideoCapture(video_path) \n        ret, frame = capture_image.read()\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        ax[i//3, i%3].imshow(frame)\n        ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n        ax[i//3, i%3].axis('on')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:33.661468Z","iopub.execute_input":"2024-08-29T09:54:33.661797Z","iopub.status.idle":"2024-08-29T09:54:33.671498Z","shell.execute_reply.started":"2024-08-29T09:54:33.661737Z","shell.execute_reply":"2024-08-29T09:54:33.670568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='atvmxvwyns.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:33.673071Z","iopub.execute_input":"2024-08-29T09:54:33.67364Z","iopub.status.idle":"2024-08-29T09:54:36.34342Z","shell.execute_reply.started":"2024-08-29T09:54:33.673344Z","shell.execute_reply":"2024-08-29T09:54:36.342659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test video files","metadata":{}},{"cell_type":"markdown","source":"1. ","metadata":{}},{"cell_type":"markdown","source":"Let's also look to few of the test data files.","metadata":{}},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:40.375989Z","iopub.execute_input":"2024-08-29T09:54:40.376377Z","iopub.status.idle":"2024-08-29T09:54:40.384163Z","shell.execute_reply.started":"2024-08-29T09:54:40.376309Z","shell.execute_reply":"2024-08-29T09:54:40.38342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:40.575744Z","iopub.execute_input":"2024-08-29T09:54:40.576051Z","iopub.status.idle":"2024-08-29T09:54:40.584777Z","shell.execute_reply.started":"2024-08-29T09:54:40.576002Z","shell.execute_reply":"2024-08-29T09:54:40.58406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualize now one of the videos.","metadata":{}},{"cell_type":"code","source":"display_image_from_video(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[2].video))","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:41.241608Z","iopub.execute_input":"2024-08-29T09:54:41.241965Z","iopub.status.idle":"2024-08-29T09:54:41.723745Z","shell.execute_reply.started":"2024-08-29T09:54:41.241915Z","shell.execute_reply":"2024-08-29T09:54:41.722798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Play video files","metadata":{}},{"cell_type":"markdown","source":"Let's look to few fake videos.","metadata":{}},{"cell_type":"code","source":"fake_videos = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].index)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:54:41.889353Z","iopub.execute_input":"2024-08-29T09:54:41.889693Z","iopub.status.idle":"2024-08-29T09:54:41.895332Z","shell.execute_reply.started":"2024-08-29T09:54:41.889633Z","shell.execute_reply":"2024-08-29T09:54:41.894356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\n\ndef play_video(video_file, subset=TRAIN_SAMPLE_FOLDER):\n    '''\n    Display video\n    param: video_file - the name of the video file to display\n    param: subset - the folder where the video file is located (can be TRAIN_SAMPLE_FOLDER or TEST_Folder)\n    '''\n    video_url = open(os.path.join(DATA_FOLDER, subset,video_file),'rb').read()\n    data_url = \"data:video/mp4;base64,\" + b64encode(video_url).decode()\n    return HTML(\"\"\"<video width=500 controls><source src=\"%s\" type=\"video/mp4\"></video>\"\"\" % data_url)\n\nplay_video(fake_videos[10])","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:17.270494Z","iopub.execute_input":"2024-08-29T09:55:17.270828Z","iopub.status.idle":"2024-08-29T09:55:17.477303Z","shell.execute_reply.started":"2024-08-29T09:55:17.270782Z","shell.execute_reply":"2024-08-29T09:55:17.475899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From visual inspection of these fakes videos, in some cases is very easy to spot the anomalies created when engineering the deep fake, in some cases is more difficult.","metadata":{}},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"markdown","source":"### A CNN-RNN Architecture","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 64\nEPOCHS = 10\n\nMAX_SEQ_LENGTH = 20\nNUM_FEATURES = 2048","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:18.19676Z","iopub.execute_input":"2024-08-29T09:55:18.197053Z","iopub.status.idle":"2024-08-29T09:55:18.201812Z","shell.execute_reply.started":"2024-08-29T09:55:18.197012Z","shell.execute_reply":"2024-08-29T09:55:18.200843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" In this example we will do the following:\n\n* Capture the frames of a video.\n* Extract frames from the videos until a maximum frame count is reached.\n* In the case, where a video's frame count is lesser than the maximum frame count we will pad the video with zeros.","metadata":{}},{"cell_type":"code","source":"def crop_center_square(frame):\n    y, x = frame.shape[0:2]\n    min_dim = min(y, x)\n    start_x = (x // 2) - (min_dim // 2)\n    start_y = (y // 2) - (min_dim // 2)\n    return frame[start_y : start_y + min_dim, start_x : start_x + min_dim]\n\n\ndef load_video(path, max_frames=0, resize=(IMG_SIZE, IMG_SIZE)):\n    cap = cv2.VideoCapture(path)\n    frames = []\n    try:\n        while True:\n            ret, frame = cap.read()\n            if not ret:\n                break\n            frame = crop_center_square(frame)\n            frame = cv2.resize(frame, resize)\n            frame = frame[:, :, [2, 1, 0]]\n            frames.append(frame)\n\n            if len(frames) == max_frames:\n                break\n    finally:\n        cap.release()\n    return np.array(frames)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:18.610938Z","iopub.execute_input":"2024-08-29T09:55:18.611271Z","iopub.status.idle":"2024-08-29T09:55:18.621535Z","shell.execute_reply.started":"2024-08-29T09:55:18.611216Z","shell.execute_reply":"2024-08-29T09:55:18.620826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can use a pre-trained network to extract meaningful features from the extracted frames. The Keras Applications module provides a number of state-of-the-art models pre-trained on the ImageNet-1k dataset. We will be using the InceptionV3 model for this purpose.","metadata":{}},{"cell_type":"code","source":"def build_feature_extractor():\n    feature_extractor = keras.applications.InceptionV3(\n        weights=\"imagenet\",\n        include_top=False,\n        pooling=\"avg\",\n        input_shape=(IMG_SIZE, IMG_SIZE, 3),\n    )\n    preprocess_input = keras.applications.inception_v3.preprocess_input\n\n    inputs = keras.Input((IMG_SIZE, IMG_SIZE, 3))\n    preprocessed = preprocess_input(inputs)\n\n    outputs = feature_extractor(preprocessed)\n    return keras.Model(inputs, outputs, name=\"feature_extractor\")\n\n\nfeature_extractor = build_feature_extractor()","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:19.084016Z","iopub.execute_input":"2024-08-29T09:55:19.084325Z","iopub.status.idle":"2024-08-29T09:55:22.71862Z","shell.execute_reply.started":"2024-08-29T09:55:19.084283Z","shell.execute_reply":"2024-08-29T09:55:22.717813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finally, we can put all the pieces together to create our data processing utility.","metadata":{}},{"cell_type":"code","source":"def prepare_all_videos(df, root_dir):\n    num_samples = len(df)\n    video_paths = list(df.index)\n    labels = df[\"label\"].values\n    labels = np.array(labels=='FAKE').astype(np.int)\n\n    # `frame_masks` and `frame_features` are what we will feed to our sequence model.\n    # `frame_masks` will contain a bunch of booleans denoting if a timestep is\n    # masked with padding or not.\n    frame_masks = np.zeros(shape=(num_samples, MAX_SEQ_LENGTH), dtype=\"bool\")\n    frame_features = np.zeros(\n        shape=(num_samples, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\"\n    )\n\n    # For each video.\n    for idx, path in enumerate(video_paths):\n        # Gather all its frames and add a batch dimension.\n        frames = load_video(os.path.join(root_dir, path))\n        frames = frames[None, ...]\n\n        # Initialize placeholders to store the masks and features of the current video.\n        temp_frame_mask = np.zeros(shape=(1, MAX_SEQ_LENGTH,), dtype=\"bool\")\n        temp_frame_features = np.zeros(\n            shape=(1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\"\n        )\n\n        # Extract features from the frames of the current video.\n        for i, batch in enumerate(frames):\n            video_length = batch.shape[0]\n            length = min(MAX_SEQ_LENGTH, video_length)\n            for j in range(length):\n                temp_frame_features[i, j, :] = feature_extractor.predict(\n                    batch[None, j, :]\n                )\n            temp_frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n        frame_features[idx,] = temp_frame_features.squeeze()\n        frame_masks[idx,] = temp_frame_mask.squeeze()\n\n    return (frame_features, frame_masks), labels","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:22.721123Z","iopub.execute_input":"2024-08-29T09:55:22.721497Z","iopub.status.idle":"2024-08-29T09:55:22.735205Z","shell.execute_reply.started":"2024-08-29T09:55:22.721434Z","shell.execute_reply":"2024-08-29T09:55:22.734419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since we don't have test labels we split the training data to find its performance in unseen data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(train_sample_metadata,test_size=0.1,random_state=42,stratify=train_sample_metadata['label'])\n\nprint(Train_set.shape, Test_set.shape )","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:22.736599Z","iopub.execute_input":"2024-08-29T09:55:22.736926Z","iopub.status.idle":"2024-08-29T09:55:23.594395Z","shell.execute_reply.started":"2024-08-29T09:55:22.736873Z","shell.execute_reply":"2024-08-29T09:55:23.593575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, train_labels = prepare_all_videos(Train_set, \"train\")\ntest_data, test_labels = prepare_all_videos(Test_set, \"test\")\n\nprint(f\"Frame features in train set: {train_data[0].shape}\")\nprint(f\"Frame masks in train set: {train_data[1].shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:23.595911Z","iopub.execute_input":"2024-08-29T09:55:23.596243Z","iopub.status.idle":"2024-08-29T09:55:23.685594Z","shell.execute_reply.started":"2024-08-29T09:55:23.596179Z","shell.execute_reply":"2024-08-29T09:55:23.68473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The sequence model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, GRU, Dense, Dropout","metadata":{"execution":{"iopub.status.busy":"2024-08-29T10:39:53.852835Z","iopub.execute_input":"2024-08-29T10:39:53.853641Z","iopub.status.idle":"2024-08-29T10:39:53.860926Z","shell.execute_reply.started":"2024-08-29T10:39:53.853375Z","shell.execute_reply":"2024-08-29T10:39:53.860066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we can feed this data to a sequence model consisting of recurrent layers like GRU.","metadata":{}},{"cell_type":"code","source":"frame_features_input = keras.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\nmask_input = keras.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n\n# Refer to the following tutorial to understand the significance of using `mask`:\n# https://keras.io/api/layers/recurrent_layers/gru/\nx = keras.layers.GRU(16, return_sequences=True)(\n    frame_features_input, mask=mask_input\n)\nx = keras.layers.GRU(8)(x)\nx = keras.layers.Dropout(0.4)(x)\nx = keras.layers.Dense(8, activation=\"relu\")(x)\noutput = keras.layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = keras.Model([frame_features_input, mask_input], output)\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:23.688197Z","iopub.execute_input":"2024-08-29T09:55:23.688542Z","iopub.status.idle":"2024-08-29T09:55:25.324905Z","shell.execute_reply.started":"2024-08-29T09:55:23.68848Z","shell.execute_reply":"2024-08-29T09:55:25.323894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = keras.callbacks.ModelCheckpoint('./', save_weights_only=True, save_best_only=True)\nhistory = model.fit(\n        [train_data[0], train_data[1]],\n        train_labels,\n        validation_data=([test_data[0], test_data[1]],test_labels),\n        callbacks=[checkpoint],\n        epochs=EPOCHS,\n        batch_size=8\n    )","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:55:25.326882Z","iopub.execute_input":"2024-08-29T09:55:25.327152Z","iopub.status.idle":"2024-08-29T09:55:40.677018Z","shell.execute_reply.started":"2024-08-29T09:55:25.327109Z","shell.execute_reply":"2024-08-29T09:55:40.67628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"### Inference","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_single_video(frames):\n    frames = frames[None, ...]\n    frame_mask = np.zeros(shape=(1, MAX_SEQ_LENGTH,), dtype=\"bool\")\n    frame_features = np.zeros(shape=(1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    for i, batch in enumerate(frames):\n        video_length = batch.shape[0]\n        length = min(MAX_SEQ_LENGTH, video_length)\n        for j in range(length):\n            frame_features[i, j, :] = feature_extractor.predict(batch[None, j, :])\n        frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n    return frame_features, frame_mask\n\ndef sequence_prediction(path):\n    frames = load_video(os.path.join(DATA_FOLDER, TEST_FOLDER,path))\n    frame_features, frame_mask = prepare_single_video(frames)\n    return model.predict([frame_features, frame_mask])[0]\n    \n# This utility is for visualization.\n# Referenced from:\n# https://www.tensorflow.org/hub/tutorials/action_recognition_with_tf_hub\ndef to_gif(images):\n    converted_images = images.astype(np.uint8)\n    imageio.mimsave(\"animation.gif\", converted_images, fps=10)\n    return embed.embed_file(\"animation.gif\")\n\n\ntest_video = np.random.choice(test_videos[\"video\"].values.tolist())\nprint(f\"Test video path: {test_video}\")\n\nif(sequence_prediction(test_video)>=0.5):\n    print(f'The predicted class of the video is FAKE')\nelse:\n    print(f'The predicted class of the video is REAL')\n\nplay_video(test_video,TEST_FOLDER)","metadata":{"execution":{"iopub.status.busy":"2024-08-29T11:11:17.494542Z","iopub.execute_input":"2024-08-29T11:11:17.494873Z","iopub.status.idle":"2024-08-29T11:11:22.305449Z","shell.execute_reply.started":"2024-08-29T11:11:17.494822Z","shell.execute_reply":"2024-08-29T11:11:22.303377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.predict('hierggamuo.mp4')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"execution":{"iopub.status.busy":"2024-08-29T10:00:07.507922Z","iopub.execute_input":"2024-08-29T10:00:07.508241Z","iopub.status.idle":"2024-08-29T10:00:07.512241Z","shell.execute_reply.started":"2024-08-29T10:00:07.5082Z","shell.execute_reply":"2024-08-29T10:00:07.511357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('/kaggle/working/rnn_model.h5')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T09:58:53.845671Z","iopub.execute_input":"2024-08-29T09:58:53.846091Z","iopub.status.idle":"2024-08-29T09:58:53.880972Z","shell.execute_reply.started":"2024-08-29T09:58:53.846027Z","shell.execute_reply":"2024-08-29T09:58:53.880115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel.load_weights('/kaggle/working/rnn_model.h5')  # Make sure this path matches where you saved your weights\n\n# Evaluate the model on the test set\ntest_loss, test_accuracy = model.evaluate([test_data[0], test_data[1]], test_labels, batch_size=8)\n\nprint(f'Test Loss: {test_loss}')\nprint(f'Test Accuracy: {test_accuracy}')\n\n# Predictions\ntest_predictions = model.predict([test_data[0], test_data[1]], batch_size=8)\ntest_predictions = (test_predictions > 0.5).astype(int)  # Thresholding for binary classification\n\n# Compute additional metrics (if needed)\nif len(test_labels.shape) == 2 and test_labels.shape[1] == 1:  # Assuming binary classification\n    test_labels = test_labels.flatten()\n    test_predictions = test_predictions.flatten()\n\nprint('Classification Report:')\nprint(classification_report(test_labels, test_predictions))\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T10:00:13.106725Z","iopub.execute_input":"2024-08-29T10:00:13.107082Z","iopub.status.idle":"2024-08-29T10:00:13.274586Z","shell.execute_reply.started":"2024-08-29T10:00:13.107019Z","shell.execute_reply":"2024-08-29T10:00:13.273651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we used simple RNN model feel free to try some complex Attention based and Transformer based models","metadata":{}},{"cell_type":"markdown","source":"All most 80% accuracy","metadata":{}},{"cell_type":"code","source":"print(model.input_shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T11:31:36.180088Z","iopub.execute_input":"2024-08-29T11:31:36.180642Z","iopub.status.idle":"2024-08-29T11:31:36.185233Z","shell.execute_reply.started":"2024-08-29T11:31:36.180387Z","shell.execute_reply":"2024-08-29T11:31:36.184436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_single_video(frames):\n    # Ensure frames shape is (batch_size, sequence_length, height, width, channels)\n    frames = frames[None, ...]  # Add batch dimension: (1, sequence_length, height, width, channels)\n    frame_mask = np.zeros(shape=(1, MAX_SEQ_LENGTH,), dtype=\"bool\")\n    frame_features = np.zeros(shape=(1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    for i, batch in enumerate(frames):\n        video_length = batch.shape[1]  # Sequence length\n        length = min(MAX_SEQ_LENGTH, video_length)\n        for j in range(length):\n            # Extract features using feature extractor\n            frame_features[i, j, :] = feature_extractor.predict(batch[j, :, :][None, ...])  # Add batch dimension for feature extractor\n        frame_mask[i, :length] = 1  # 1 = not masked, 0 = masked\n\n    return frame_features, frame_mask\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T11:31:42.560758Z","iopub.execute_input":"2024-08-29T11:31:42.56119Z","iopub.status.idle":"2024-08-29T11:31:42.573169Z","shell.execute_reply.started":"2024-08-29T11:31:42.561126Z","shell.execute_reply":"2024-08-29T11:31:42.572175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=(MAX_SEQ_LENGTH, 224, 224, 3)),  # Adjust input shape\n        tf.keras.layers.GRU(128, return_sequences=True),\n        tf.keras.layers.Dense(1, activation=\"sigmoid\")\n    ])\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T11:31:49.706456Z","iopub.execute_input":"2024-08-29T11:31:49.706836Z","iopub.status.idle":"2024-08-29T11:31:49.713077Z","shell.execute_reply.started":"2024-08-29T11:31:49.706773Z","shell.execute_reply":"2024-08-29T11:31:49.712126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sequence_prediction(video_path):\n    frames = load_video(video_path)\n    frame_features, frame_mask = prepare_single_video(frames)\n    adjusted_model = build_model()\n    # Load weights into the model\n    adjusted_model.load_weights('rnn_model.h5')\n    return adjusted_model.predict([frame_features, frame_mask])[0]\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T11:32:04.165112Z","iopub.execute_input":"2024-08-29T11:32:04.16546Z","iopub.status.idle":"2024-08-29T11:32:04.171101Z","shell.execute_reply.started":"2024-08-29T11:32:04.165405Z","shell.execute_reply":"2024-08-29T11:32:04.170173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('/kaggle/working/rnn_model.h5')\n","metadata":{"execution":{"iopub.status.busy":"2024-08-29T11:32:27.252527Z","iopub.execute_input":"2024-08-29T11:32:27.252853Z","iopub.status.idle":"2024-08-29T11:32:27.274506Z","shell.execute_reply.started":"2024-08-29T11:32:27.252808Z","shell.execute_reply":"2024-08-29T11:32:27.273859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}