{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":16880,"databundleVersionId":858837}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# from tensorflow_docs.vis import embed\nfrom tensorflow import keras\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-23T05:17:47.750373Z","iopub.execute_input":"2024-04-23T05:17:47.750872Z","iopub.status.idle":"2024-04-23T05:17:53.662558Z","shell.execute_reply.started":"2024-04-23T05:17:47.750791Z","shell.execute_reply":"2024-04-23T05:17:53.661337Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:53.664887Z","iopub.execute_input":"2024-04-23T05:17:53.665272Z","iopub.status.idle":"2024-04-23T05:17:53.981885Z","shell.execute_reply.started":"2024-04-23T05:17:53.665200Z","shell.execute_reply":"2024-04-23T05:17:53.980814Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:53.983557Z","iopub.execute_input":"2024-04-23T05:17:53.983930Z","iopub.status.idle":"2024-04-23T05:17:54.383215Z","shell.execute_reply.started":"2024-04-23T05:17:53.983865Z","shell.execute_reply":"2024-04-23T05:17:54.382158Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:54.384945Z","iopub.execute_input":"2024-04-23T05:17:54.385328Z","iopub.status.idle":"2024-04-23T05:17:54.704131Z","shell.execute_reply.started":"2024-04-23T05:17:54.385255Z","shell.execute_reply":"2024-04-23T05:17:54.702440Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:54.710485Z","iopub.execute_input":"2024-04-23T05:17:54.711108Z","iopub.status.idle":"2024-04-23T05:17:54.720272Z","shell.execute_reply.started":"2024-04-23T05:17:54.711013Z","shell.execute_reply":"2024-04-23T05:17:54.719272Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(10).index)\nfake_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:54.724136Z","iopub.execute_input":"2024-04-23T05:17:54.724740Z","iopub.status.idle":"2024-04-23T05:17:54.738024Z","shell.execute_reply.started":"2024-04-23T05:17:54.724672Z","shell.execute_reply":"2024-04-23T05:17:54.737129Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef show_first_frame(video_file_path):\n    \"\"\"\n    Fetches and displays the first frame of a given video.\n    \n    Parameters:\n        video_file_path (str): The full path to the video file.\n    \n    Raises:\n        FileNotFoundError: Raised if the video file cannot be found or opened.\n        RuntimeError: Raised if the video file contains no readable frames.\n    \"\"\"\n    # Set up the video capture object to read from the video file\n    video_capture = cv2.VideoCapture(video_file_path)\n    \n    # Verify that the video file was opened successfully\n    if not video_capture.isOpened():\n        video_capture.release()  # Ensure resources are released\n        raise FileNotFoundError(f\"Failed to access the video at {video_file_path}\")\n    \n    # Attempt to capture the first frame\n    successful, frame = video_capture.read()\n    \n    # Ensure that a frame was successfully captured\n    if not successful:\n        video_capture.release()  # Ensure resources are released before raising an error\n        raise RuntimeError(\"No frames could be read from the video file\")\n    \n    # Adjust the frame's color format for displaying\n    frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n    \n    # Display the frame using matplotlib\n    plt.figure(figsize=(10, 10))\n    plt.imshow(frame_rgb)\n    plt.axis('off')  # Hide axes for better visualization\n    plt.show()\n    \n    # Close the video capture object to free resources\n    video_capture.release()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:54.740145Z","iopub.execute_input":"2024-04-23T05:17:54.740594Z","iopub.status.idle":"2024-04-23T05:17:54.752861Z","shell.execute_reply.started":"2024-04-23T05:17:54.740524Z","shell.execute_reply":"2024-04-23T05:17:54.751955Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    show_first_frame(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:54.754143Z","iopub.execute_input":"2024-04-23T05:17:54.754557Z","iopub.status.idle":"2024-04-23T05:17:58.962677Z","shell.execute_reply.started":"2024-04-23T05:17:54.754513Z","shell.execute_reply":"2024-04-23T05:17:58.961562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(10).index)\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:58.964390Z","iopub.execute_input":"2024-04-23T05:17:58.964762Z","iopub.status.idle":"2024-04-23T05:17:58.975331Z","shell.execute_reply.started":"2024-04-23T05:17:58.964702Z","shell.execute_reply":"2024-04-23T05:17:58.974336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    show_first_frame(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:17:58.977040Z","iopub.execute_input":"2024-04-23T05:17:58.977566Z","iopub.status.idle":"2024-04-23T05:18:03.477798Z","shell.execute_reply.started":"2024-04-23T05:17:58.977491Z","shell.execute_reply":"2024-04-23T05:18:03.476961Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata['original'].value_counts()[0:10]","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:03.479508Z","iopub.execute_input":"2024-04-23T05:18:03.479861Z","iopub.status.idle":"2024-04-23T05:18:03.489941Z","shell.execute_reply.started":"2024-04-23T05:18:03.479802Z","shell.execute_reply":"2024-04-23T05:18:03.489073Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\n\ndef show_frames_from_videos(video_files, base_folder=TRAIN_SAMPLE_FOLDER):\n    \"\"\"\n    Displays the first frame from each video in the given list of video files.\n    \n    Parameters:\n        video_files (list): A list of video file names.\n        base_folder (str): The directory containing the video files, default is TRAIN_SAMPLE_FOLDER.\n    \n    Description:\n        This function plots the first frame of the first six videos from the provided list. Each frame is displayed\n        in a grid format using matplotlib.\n    \"\"\"\n    # Create a figure with subplots arranged in 2 rows and 3 columns\n    fig, axes = plt.subplots(2, 3, figsize=(16, 8))\n    \n    # Loop through the first six videos in the list\n    for index, video_name in enumerate(video_files[:6]):\n        video_full_path = os.path.join(DATA_FOLDER, base_folder, video_name)\n        video_capture = cv2.VideoCapture(video_full_path)\n        \n        success, frame = video_capture.read()\n        if not success:\n            print(f\"Failed to read from {video_name}\")\n            axes[index // 3, index % 3].set_title(\"Failed to load video\")\n            axes[index // 3, index % 3].axis('off')\n            continue\n        \n        # Convert the color from BGR to RGB\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        \n        # Display the image in the respective subplot\n        axes[index // 3, index % 3].imshow(frame_rgb)\n        axes[index // 3, index % 3].set_title(video_name)\n        axes[index // 3, index % 3].axis('on')  # Keep the axis on for clarity\n        \n        # Release the video capture object\n        video_capture.release()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:03.491698Z","iopub.execute_input":"2024-04-23T05:18:03.492057Z","iopub.status.idle":"2024-04-23T05:18:03.506974Z","shell.execute_reply.started":"2024-04-23T05:18:03.491993Z","shell.execute_reply":"2024-04-23T05:18:03.505940Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='meawmsgiti.mp4'].index)\nshow_frames_from_videos(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:03.508474Z","iopub.execute_input":"2024-04-23T05:18:03.508827Z","iopub.status.idle":"2024-04-23T05:18:06.645277Z","shell.execute_reply.started":"2024-04-23T05:18:03.508760Z","shell.execute_reply":"2024-04-23T05:18:06.644315Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:06.646929Z","iopub.execute_input":"2024-04-23T05:18:06.647292Z","iopub.status.idle":"2024-04-23T05:18:06.655632Z","shell.execute_reply.started":"2024-04-23T05:18:06.647226Z","shell.execute_reply":"2024-04-23T05:18:06.654527Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_videos.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:06.657836Z","iopub.execute_input":"2024-04-23T05:18:06.658320Z","iopub.status.idle":"2024-04-23T05:18:06.674067Z","shell.execute_reply.started":"2024-04-23T05:18:06.658231Z","shell.execute_reply":"2024-04-23T05:18:06.672833Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_first_frame(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[3].video))","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:06.675700Z","iopub.execute_input":"2024-04-23T05:18:06.676006Z","iopub.status.idle":"2024-04-23T05:18:07.086842Z","shell.execute_reply.started":"2024-04-23T05:18:06.675951Z","shell.execute_reply":"2024-04-23T05:18:07.085771Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fake_videos = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].index)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:07.088527Z","iopub.execute_input":"2024-04-23T05:18:07.088951Z","iopub.status.idle":"2024-04-23T05:18:07.095827Z","shell.execute_reply.started":"2024-04-23T05:18:07.088882Z","shell.execute_reply":"2024-04-23T05:18:07.094872Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\nimport os\n\ndef embed_video_in_notebook(video_filename, directory=TRAIN_SAMPLE_FOLDER):\n    \"\"\"\n    Embeds a video file into an IPython notebook for playback.\n\n    Args:\n        video_filename (str): The filename of the video to embed.\n        directory (str): The directory where the video file is located. Default is TRAIN_SAMPLE_FOLDER.\n\n    Returns:\n        HTML: An HTML object that embeds a video player within the notebook.\n    \n    Raises:\n        FileNotFoundError: If the video file cannot be found in the specified directory.\n    \"\"\"\n    try:\n        # Construct the full path to the video file\n        video_path = os.path.join(DATA_FOLDER, directory, video_filename)\n        \n        # Read the video file as binary data\n        with open(video_path, 'rb') as video_file:\n            video_data = video_file.read()\n\n        # Encode the video data in base64 and create the data URL\n        video_base64 = b64encode(video_data).decode('utf-8')\n        data_url = f\"data:video/mp4;base64,{video_base64}\"\n\n        # Return an HTML object that contains the video element\n        return HTML(f'<video width=\"500\" controls><source src=\"{data_url}\" type=\"video/mp4\"></video>')\n    \n    except FileNotFoundError:\n        raise FileNotFoundError(f\"The video file {video_filename} could not be found in {directory}.\")\n\n# Example usage:\n# Assuming 'fake_videos[10]' contains the filename of the video to play\nvideo_to_play = fake_videos[14] \nembed_video_in_notebook(video_to_play)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:07.097865Z","iopub.execute_input":"2024-04-23T05:18:07.098420Z","iopub.status.idle":"2024-04-23T05:18:07.265542Z","shell.execute_reply.started":"2024-04-23T05:18:07.098191Z","shell.execute_reply":"2024-04-23T05:18:07.264148Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 64\nEPOCHS = 20\n\nMAX_SEQ_LENGTH = 20\nNUM_FEATURES = 2048","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:07.267898Z","iopub.execute_input":"2024-04-23T05:18:07.268421Z","iopub.status.idle":"2024-04-23T05:18:07.273885Z","shell.execute_reply.started":"2024-04-23T05:18:07.268329Z","shell.execute_reply":"2024-04-23T05:18:07.272900Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef square_crop_frame(image):\n    \"\"\"\n    Crops the given image to a square format by reducing the largest dimension to match the smallest one.\n    \n    Parameters:\n        image (numpy.ndarray): The frame to be cropped, assumed to be in height x width x channels format.\n    \n    Returns:\n        numpy.ndarray: A square-cropped version of the input image.\n    \"\"\"\n    height, width = image.shape[:2]\n    min_dimension = min(height, width)\n    start_x = (width - min_dimension) // 2\n    start_y = (height - min_dimension) // 2\n    return image[start_y:start_y + min_dimension, start_x:start_x + min_dimension]\n\ndef process_video_frames(video_path, max_frames=0, resize_dims=(IMG_SIZE, IMG_SIZE)):\n    \"\"\"\n    Extracts and processes frames from a video file, resizing them and converting color format.\n\n    Parameters:\n        video_path (str): Full path to the video file.\n        max_frames (int): Maximum number of frames to extract. Extracts all if set to zero.\n        resize_dims (tuple): Dimensions to resize frames to, in (width, height) format.\n\n    Returns:\n        numpy.ndarray: An array of processed video frames.\n    \"\"\"\n    capture = cv2.VideoCapture(video_path)\n    processed_frames = []\n    try:\n        while True:\n            read_success, frame = capture.read()\n            if not read_success:\n                break\n            frame = square_crop_frame(frame)\n            frame = cv2.resize(frame, resize_dims)\n            # Convert BGR to RGB for standard color format\n            frame = frame[..., ::-1]\n            processed_frames.append(frame)\n\n            if max_frames > 0 and len(processed_frames) >= max_frames:\n                break\n    finally:\n        capture.release()\n    return np.array(processed_frames)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:07.275486Z","iopub.execute_input":"2024-04-23T05:18:07.275820Z","iopub.status.idle":"2024-04-23T05:18:07.290361Z","shell.execute_reply.started":"2024-04-23T05:18:07.275760Z","shell.execute_reply":"2024-04-23T05:18:07.289179Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\ndef build_feature_extractor(model_name='ResNet50'):\n    \"\"\"\n    Builds a feature extraction model using a specified base model from Keras Applications.\n\n    Args:\n        model_name (str): Name of the model to use ('ResNet50', 'VGG16', 'MobileNetV2', etc.).\n\n    Returns:\n        keras.Model: A Keras model that takes an image as input and outputs the extracted features.\n    \"\"\"\n    # Dynamically retrieving the model and preprocessing function based on the model_name\n    base_model_class = getattr(keras.applications, model_name)\n    base_model = base_model_class(\n        weights=\"imagenet\",\n        include_top=False,\n        pooling=\"avg\",\n        input_shape=(IMG_SIZE, IMG_SIZE, 3)\n    )\n    preprocess_input = getattr(keras.applications, model_name.lower()).preprocess_input\n\n    # Define the input layer\n    inputs = keras.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n    # Preprocess input\n    x = preprocess_input(inputs)\n    # Get features\n    outputs = base_model(x)\n\n    # Create the model\n    model = keras.Model(inputs=inputs, outputs=outputs, name=f\"{model_name}_feature_extractor\")\n    \n    return model\n\n# Example usage:\nfeature_extractor = build_feature_extractor('ResNet50')  # I also tried with 'ResNet50', 'VGG16', etc.","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:07.292835Z","iopub.execute_input":"2024-04-23T05:18:07.293350Z","iopub.status.idle":"2024-04-23T05:18:12.785991Z","shell.execute_reply.started":"2024-04-23T05:18:07.293261Z","shell.execute_reply":"2024-04-23T05:18:12.784886Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport os\n\ndef extract_video_features(dataframe, directory):\n    \"\"\"\n    Processes all videos specified in a dataframe to extract features for model input, including masks.\n\n    Args:\n        dataframe (pd.DataFrame): DataFrame containing indices as video file paths and a 'label' column.\n        directory (str): Base directory where video files are stored.\n\n    Returns:\n        tuple: A tuple (features_and_masks, labels) where features_and_masks contains the video\n               features and masks indicating valid data points, and labels are the binary labels.\n    \"\"\"\n    total_videos = len(dataframe)\n    video_file_paths = dataframe.index.tolist()\n    binary_labels = np.array(dataframe[\"label\"].values == 'FAKE', dtype=int)\n\n    # Initialize arrays to hold data for all videos\n    video_masks = np.zeros((total_videos, MAX_SEQ_LENGTH), dtype=bool)\n    video_features = np.zeros((total_videos, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    # Process each video individually\n    for video_idx, video_file in enumerate(video_file_paths):\n        full_video_path = os.path.join(directory, video_file)\n        video_data = process_video_frames(full_video_path)\n        video_data = np.expand_dims(video_data, axis=0)  # Add a batch dimension\n\n        # Temporary storage for this video's data\n        current_video_mask = np.zeros((1, MAX_SEQ_LENGTH), dtype=bool)\n        current_video_features = np.zeros((1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n        # Frame-by-frame feature extraction\n        frames_to_process = min(MAX_SEQ_LENGTH, video_data.shape[1])\n        for frame_idx in range(frames_to_process):\n            frame = video_data[:, frame_idx, :]\n            extracted_features = feature_extractor.predict(frame[None, :])\n            current_video_features[0, frame_idx, :] = extracted_features\n\n        current_video_mask[0, :frames_to_process] = True  # Mark frames as valid\n\n        # Store the extracted data in the corresponding arrays\n        video_features[video_idx] = current_video_features.squeeze()\n        video_masks[video_idx] = current_video_mask.squeeze()\n\n    return (video_features, video_masks), binary_labels","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:12.788021Z","iopub.execute_input":"2024-04-23T05:18:12.788421Z","iopub.status.idle":"2024-04-23T05:18:12.805395Z","shell.execute_reply.started":"2024-04-23T05:18:12.788352Z","shell.execute_reply":"2024-04-23T05:18:12.804044Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(train_sample_metadata,test_size=0.1,random_state=42,stratify=train_sample_metadata['label'])\n\nprint(Train_set.shape, Test_set.shape )","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:12.807069Z","iopub.execute_input":"2024-04-23T05:18:12.807384Z","iopub.status.idle":"2024-04-23T05:18:12.848576Z","shell.execute_reply.started":"2024-04-23T05:18:12.807326Z","shell.execute_reply":"2024-04-23T05:18:12.847523Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, train_labels = extract_video_features(Train_set, \"train\")\ntest_data, test_labels = extract_video_features(Test_set, \"test\")\n\nprint(f\"Frame features in train set: {train_data[0].shape}\")\nprint(f\"Frame masks in train set: {train_data[1].shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:12.850440Z","iopub.execute_input":"2024-04-23T05:18:12.850736Z","iopub.status.idle":"2024-04-23T05:18:12.948090Z","shell.execute_reply.started":"2024-04-23T05:18:12.850687Z","shell.execute_reply":"2024-04-23T05:18:12.946869Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models, regularizers\n\nframe_features_input = layers.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\nmask_input = layers.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n\nx = layers.Bidirectional(layers.GRU(16, return_sequences=True, kernel_regularizer=regularizers.l2(0.01)))(\n    frame_features_input, mask=mask_input\n)\nx = layers.Bidirectional(layers.GRU(8, kernel_regularizer=regularizers.l2(0.01)))(x)\nx = layers.Dropout(0.5)(x)  # Increased dropout\nx = layers.Dense(8, activation=\"relu\")(x)\noutput = layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = models.Model([frame_features_input, mask_input], output)\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:12.949444Z","iopub.execute_input":"2024-04-23T05:18:12.949795Z","iopub.status.idle":"2024-04-23T05:18:16.124294Z","shell.execute_reply.started":"2024-04-23T05:18:12.949734Z","shell.execute_reply":"2024-04-23T05:18:16.123269Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tensorflow.keras import callbacks, models\n\n# Define the directory for storing model checkpoints and the final model\ncheckpoint_dir = './model_checkpoints'\nos.makedirs(checkpoint_dir, exist_ok=True)  # Ensure the directory exists\n\n# Setup the model checkpoint callback to save only the best model during training\ncheckpoint_filepath = os.path.join(checkpoint_dir, 'model-{epoch:02d}-{val_loss:.2f}.h5')\ncheckpoint_callback = callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    monitor='val_loss',\n    verbose=1,\n    save_best_only=True,\n    save_weights_only=True,\n    mode='min'\n)\n\n# EarlyStopping callback to stop training early if no improvement\nearly_stopping_callback = callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    verbose=1,\n    mode='min',\n    restore_best_weights=True\n)\n\n# Model training\nhistory = model.fit(\n    [train_data[0], train_data[1]],\n    train_labels,\n    validation_data=([test_data[0], test_data[1]], test_labels),\n    epochs=10,\n    batch_size=8,\n    callbacks=[checkpoint_callback, early_stopping_callback],\n    verbose=1\n)\n\n# Save the final model after training\nfinal_model_path = os.path.join(checkpoint_dir, 'final_model6.h5')\nmodel.save(final_model_path)\nprint(f\"Model saved to {final_model_path}\")\n\n# Optionally, print the history of training\nprint(\"Training history:\", history.history)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:16.125955Z","iopub.execute_input":"2024-04-23T05:18:16.126322Z","iopub.status.idle":"2024-04-23T05:18:43.439789Z","shell.execute_reply.started":"2024-04-23T05:18:16.126265Z","shell.execute_reply":"2024-04-23T05:18:43.438929Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model on the test set\ntest_loss, test_accuracy = model.evaluate(\n    [test_data[0], test_data[1]],  # Test features and masks\n    test_labels,                  # Test labels\n    batch_size=8                  # Using the same batch size as during training\n)\n\nprint(f\"Test Loss: {test_loss}\")\nprint(f\"Test Accuracy: {test_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:43.441264Z","iopub.execute_input":"2024-04-23T05:18:43.441583Z","iopub.status.idle":"2024-04-23T05:18:43.537681Z","shell.execute_reply.started":"2024-04-23T05:18:43.441523Z","shell.execute_reply":"2024-04-23T05:18:43.536843Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_single_video(frames):\n    frames = frames[None, ...]  # Add batch dimension if not present\n    frame_mask = np.zeros((1, MAX_SEQ_LENGTH), dtype=bool)\n    frame_features = np.zeros((1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype='float32')\n\n    video_length = frames.shape[1]\n    length = min(MAX_SEQ_LENGTH, video_length)\n\n    for j in range(length):\n        frame = np.expand_dims(frames[0, j], axis=0)  # Add batch dimension\n        features = feature_extractor.predict(frame)\n        frame_features[0, j, :] = features\n    \n    frame_mask[0, :length] = True  # Mark the frames that have data\n    return frame_features, frame_mask\n\n\ndef sequence_prediction(video_path):\n    frames = process_video_frames(video_path)\n    frame_features, frame_mask = prepare_single_video(frames)\n    prediction = model.predict([frame_features, frame_mask])[0]\n    return prediction\n\n\nimport imageio\nimport IPython.display as display\n\ndef to_gif(images):\n    imageio.mimsave('animation.gif', images, fps=10)\n    return display.Image(filename='animation.gif')\n\n\ntest_video_path = os.path.join(DATA_FOLDER, TEST_FOLDER, np.random.choice(test_videos[\"video\"].values.tolist()))\nprint(f\"Test video path: {test_video_path}\")\n\nprediction = sequence_prediction(test_video_path)\nif prediction >= 0.5:\n    print('The predicted class of the video is FAKE')\nelse:\n    print('The predicted class of the video is REAL')\n\n# Optional: Load frames again to create a GIF for visualization\nframes = process_video_frames(test_video_path)\nto_gif(frames)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:18:43.538990Z","iopub.execute_input":"2024-04-23T05:18:43.539327Z","iopub.status.idle":"2024-04-23T05:19:09.700044Z","shell.execute_reply.started":"2024-04-23T05:18:43.539266Z","shell.execute_reply":"2024-04-23T05:19:09.698822Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\n# Predict probabilities for the test set\nprobabilities = model.predict([test_data[0], test_data[1]])\nprobabilities = probabilities.flatten()  # Ensure the shape matches the labels array if necessary\n\n# Compute ROC curve and ROC area for each class\nfpr, tpr, _ = roc_curve(test_labels, probabilities)\nroc_auc = auc(fpr, tpr)\n\n# Plot ROC curve\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic example')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:19:09.701787Z","iopub.execute_input":"2024-04-23T05:19:09.702175Z","iopub.status.idle":"2024-04-23T05:19:10.081379Z","shell.execute_reply.started":"2024-04-23T05:19:09.702110Z","shell.execute_reply":"2024-04-23T05:19:10.079942Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom sklearn.metrics import classification_report\n\n# Predict class labels for the test set\npredicted_labels = model.predict([test_data[0], test_data[1]])\npredicted_labels = (predicted_labels > 0.5).astype('int').flatten()  \n\n# Generating a classification report\nreport = classification_report(test_labels, predicted_labels, target_names=['Real', 'Fake'])\n\nprint(\"Classification Report:\\n\", report)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T05:29:22.663351Z","iopub.execute_input":"2024-04-23T05:29:22.663811Z","iopub.status.idle":"2024-04-23T05:29:22.753507Z","shell.execute_reply.started":"2024-04-23T05:29:22.663736Z","shell.execute_reply":"2024-04-23T05:29:22.752376Z"},"trusted":true},"outputs":[],"execution_count":null}]}