{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# from tensorflow_docs.vis import embed\nfrom tensorflow import keras\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-21T09:37:21.977635Z","iopub.execute_input":"2024-04-21T09:37:21.978093Z","iopub.status.idle":"2024-04-21T09:37:29.209805Z","shell.execute_reply.started":"2024-04-21T09:37:21.978016Z","shell.execute_reply":"2024-04-21T09:37:29.208451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:37:34.627853Z","iopub.execute_input":"2024-04-21T09:37:34.628346Z","iopub.status.idle":"2024-04-21T09:37:34.897709Z","shell.execute_reply.started":"2024-04-21T09:37:34.628266Z","shell.execute_reply":"2024-04-21T09:37:34.896378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:37:38.49279Z","iopub.execute_input":"2024-04-21T09:37:38.493243Z","iopub.status.idle":"2024-04-21T09:37:38.901606Z","shell.execute_reply.started":"2024-04-21T09:37:38.493175Z","shell.execute_reply":"2024-04-21T09:37:38.90025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:37:42.166586Z","iopub.execute_input":"2024-04-21T09:37:42.167004Z","iopub.status.idle":"2024-04-21T09:37:42.497872Z","shell.execute_reply.started":"2024-04-21T09:37:42.166945Z","shell.execute_reply":"2024-04-21T09:37:42.496351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:37:53.864457Z","iopub.execute_input":"2024-04-21T09:37:53.865109Z","iopub.status.idle":"2024-04-21T09:37:53.873484Z","shell.execute_reply.started":"2024-04-21T09:37:53.865018Z","shell.execute_reply":"2024-04-21T09:37:53.872411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(10).index)\nfake_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:37:56.407376Z","iopub.execute_input":"2024-04-21T09:37:56.407896Z","iopub.status.idle":"2024-04-21T09:37:56.42078Z","shell.execute_reply.started":"2024-04-21T09:37:56.407847Z","shell.execute_reply":"2024-04-21T09:37:56.419731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef show_first_frame(video_file_path):\n    \"\"\"\n    Fetches and displays the first frame of a given video.\n    \n    Parameters:\n        video_file_path (str): The full path to the video file.\n    \n    Raises:\n        FileNotFoundError: Raised if the video file cannot be found or opened.\n        RuntimeError: Raised if the video file contains no readable frames.\n    \"\"\"\n    # Set up the video capture object to read from the video file\n    video_capture = cv2.VideoCapture(video_file_path)\n    \n    # Verify that the video file was opened successfully\n    if not video_capture.isOpened():\n        video_capture.release()  # Ensure resources are released\n        raise FileNotFoundError(f\"Failed to access the video at {video_file_path}\")\n    \n    # Attempt to capture the first frame\n    successful, frame = video_capture.read()\n    \n    # Ensure that a frame was successfully captured\n    if not successful:\n        video_capture.release()  # Ensure resources are released before raising an error\n        raise RuntimeError(\"No frames could be read from the video file\")\n    \n    # Adjust the frame's color format for displaying\n    frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n    \n    # Display the frame using matplotlib\n    plt.figure(figsize=(10, 10))\n    plt.imshow(frame_rgb)\n    plt.axis('off')  # Hide axes for better visualization\n    plt.show()\n    \n    # Close the video capture object to free resources\n    video_capture.release()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:38:10.038428Z","iopub.execute_input":"2024-04-21T09:38:10.038942Z","iopub.status.idle":"2024-04-21T09:38:10.049663Z","shell.execute_reply.started":"2024-04-21T09:38:10.03888Z","shell.execute_reply":"2024-04-21T09:38:10.048402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    show_first_frame(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:38:14.476788Z","iopub.execute_input":"2024-04-21T09:38:14.477166Z","iopub.status.idle":"2024-04-21T09:38:19.090688Z","shell.execute_reply.started":"2024-04-21T09:38:14.477104Z","shell.execute_reply":"2024-04-21T09:38:19.089518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(10).index)\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:39:02.659184Z","iopub.execute_input":"2024-04-21T09:39:02.659581Z","iopub.status.idle":"2024-04-21T09:39:02.671149Z","shell.execute_reply.started":"2024-04-21T09:39:02.659524Z","shell.execute_reply":"2024-04-21T09:39:02.669488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    show_first_frame(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:39:20.398283Z","iopub.execute_input":"2024-04-21T09:39:20.398699Z","iopub.status.idle":"2024-04-21T09:39:25.020034Z","shell.execute_reply.started":"2024-04-21T09:39:20.398639Z","shell.execute_reply":"2024-04-21T09:39:25.018639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sample_metadata['original'].value_counts()[0:10]","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:39:43.003499Z","iopub.execute_input":"2024-04-21T09:39:43.00387Z","iopub.status.idle":"2024-04-21T09:39:43.016916Z","shell.execute_reply.started":"2024-04-21T09:39:43.003813Z","shell.execute_reply":"2024-04-21T09:39:43.01562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\n\ndef show_frames_from_videos(video_files, base_folder=TRAIN_SAMPLE_FOLDER):\n    \"\"\"\n    Displays the first frame from each video in the given list of video files.\n    \n    Parameters:\n        video_files (list): A list of video file names.\n        base_folder (str): The directory containing the video files, default is TRAIN_SAMPLE_FOLDER.\n    \n    Description:\n        This function plots the first frame of the first six videos from the provided list. Each frame is displayed\n        in a grid format using matplotlib.\n    \"\"\"\n    # Create a figure with subplots arranged in 2 rows and 3 columns\n    fig, axes = plt.subplots(2, 3, figsize=(16, 8))\n    \n    # Loop through the first six videos in the list\n    for index, video_name in enumerate(video_files[:6]):\n        video_full_path = os.path.join(DATA_FOLDER, base_folder, video_name)\n        video_capture = cv2.VideoCapture(video_full_path)\n        \n        success, frame = video_capture.read()\n        if not success:\n            print(f\"Failed to read from {video_name}\")\n            axes[index // 3, index % 3].set_title(\"Failed to load video\")\n            axes[index // 3, index % 3].axis('off')\n            continue\n        \n        # Convert the color from BGR to RGB\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        \n        # Display the image in the respective subplot\n        axes[index // 3, index % 3].imshow(frame_rgb)\n        axes[index // 3, index % 3].set_title(video_name)\n        axes[index // 3, index % 3].axis('on')  # Keep the axis on for clarity\n        \n        # Release the video capture object\n        video_capture.release()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:40:46.976582Z","iopub.execute_input":"2024-04-21T09:40:46.976988Z","iopub.status.idle":"2024-04-21T09:40:46.99295Z","shell.execute_reply.started":"2024-04-21T09:40:46.976929Z","shell.execute_reply":"2024-04-21T09:40:46.991579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='meawmsgiti.mp4'].index)\nshow_frames_from_videos(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:41:14.172003Z","iopub.execute_input":"2024-04-21T09:41:14.172375Z","iopub.status.idle":"2024-04-21T09:41:17.361516Z","shell.execute_reply.started":"2024-04-21T09:41:14.172324Z","shell.execute_reply":"2024-04-21T09:41:17.360158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:41:36.874813Z","iopub.execute_input":"2024-04-21T09:41:36.87531Z","iopub.status.idle":"2024-04-21T09:41:36.884856Z","shell.execute_reply.started":"2024-04-21T09:41:36.875232Z","shell.execute_reply":"2024-04-21T09:41:36.883619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:41:47.493876Z","iopub.execute_input":"2024-04-21T09:41:47.494572Z","iopub.status.idle":"2024-04-21T09:41:47.50798Z","shell.execute_reply.started":"2024-04-21T09:41:47.494494Z","shell.execute_reply":"2024-04-21T09:41:47.50713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_first_frame(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[3].video))","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:42:00.839215Z","iopub.execute_input":"2024-04-21T09:42:00.839702Z","iopub.status.idle":"2024-04-21T09:42:01.259641Z","shell.execute_reply.started":"2024-04-21T09:42:00.839627Z","shell.execute_reply":"2024-04-21T09:42:01.258478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_videos = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].index)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:42:13.231826Z","iopub.execute_input":"2024-04-21T09:42:13.232408Z","iopub.status.idle":"2024-04-21T09:42:13.239833Z","shell.execute_reply.started":"2024-04-21T09:42:13.232354Z","shell.execute_reply":"2024-04-21T09:42:13.238827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\nimport os\n\ndef embed_video_in_notebook(video_filename, directory=TRAIN_SAMPLE_FOLDER):\n    \"\"\"\n    Embeds a video file into an IPython notebook for playback.\n\n    Args:\n        video_filename (str): The filename of the video to embed.\n        directory (str): The directory where the video file is located. Default is TRAIN_SAMPLE_FOLDER.\n\n    Returns:\n        HTML: An HTML object that embeds a video player within the notebook.\n    \n    Raises:\n        FileNotFoundError: If the video file cannot be found in the specified directory.\n    \"\"\"\n    try:\n        # Construct the full path to the video file\n        video_path = os.path.join(DATA_FOLDER, directory, video_filename)\n        \n        # Read the video file as binary data\n        with open(video_path, 'rb') as video_file:\n            video_data = video_file.read()\n\n        # Encode the video data in base64 and create the data URL\n        video_base64 = b64encode(video_data).decode('utf-8')\n        data_url = f\"data:video/mp4;base64,{video_base64}\"\n\n        # Return an HTML object that contains the video element\n        return HTML(f'<video width=\"500\" controls><source src=\"{data_url}\" type=\"video/mp4\"></video>')\n    \n    except FileNotFoundError:\n        raise FileNotFoundError(f\"The video file {video_filename} could not be found in {directory}.\")\n\n# Example usage:\n# Assuming 'fake_videos[10]' contains the filename of the video to play\nvideo_to_play = fake_videos[14] \nembed_video_in_notebook(video_to_play)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:42:59.163922Z","iopub.execute_input":"2024-04-21T09:42:59.164384Z","iopub.status.idle":"2024-04-21T09:42:59.324469Z","shell.execute_reply.started":"2024-04-21T09:42:59.164312Z","shell.execute_reply":"2024-04-21T09:42:59.322601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 64\nEPOCHS = 20\n\nMAX_SEQ_LENGTH = 20\nNUM_FEATURES = 2048","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:43:10.966521Z","iopub.execute_input":"2024-04-21T09:43:10.966961Z","iopub.status.idle":"2024-04-21T09:43:10.972493Z","shell.execute_reply.started":"2024-04-21T09:43:10.966887Z","shell.execute_reply":"2024-04-21T09:43:10.971501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef square_crop_frame(image):\n    \"\"\"\n    Crops the given image to a square format by reducing the largest dimension to match the smallest one.\n    \n    Parameters:\n        image (numpy.ndarray): The frame to be cropped, assumed to be in height x width x channels format.\n    \n    Returns:\n        numpy.ndarray: A square-cropped version of the input image.\n    \"\"\"\n    height, width = image.shape[:2]\n    min_dimension = min(height, width)\n    start_x = (width - min_dimension) // 2\n    start_y = (height - min_dimension) // 2\n    return image[start_y:start_y + min_dimension, start_x:start_x + min_dimension]\n\ndef process_video_frames(video_path, max_frames=0, resize_dims=(IMG_SIZE, IMG_SIZE)):\n    \"\"\"\n    Extracts and processes frames from a video file, resizing them and converting color format.\n\n    Parameters:\n        video_path (str): Full path to the video file.\n        max_frames (int): Maximum number of frames to extract. Extracts all if set to zero.\n        resize_dims (tuple): Dimensions to resize frames to, in (width, height) format.\n\n    Returns:\n        numpy.ndarray: An array of processed video frames.\n    \"\"\"\n    capture = cv2.VideoCapture(video_path)\n    processed_frames = []\n    try:\n        while True:\n            read_success, frame = capture.read()\n            if not read_success:\n                break\n            frame = square_crop_frame(frame)\n            frame = cv2.resize(frame, resize_dims)\n            # Convert BGR to RGB for standard color format\n            frame = frame[..., ::-1]\n            processed_frames.append(frame)\n\n            if max_frames > 0 and len(processed_frames) >= max_frames:\n                break\n    finally:\n        capture.release()\n    return np.array(processed_frames)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:43:31.224539Z","iopub.execute_input":"2024-04-21T09:43:31.224955Z","iopub.status.idle":"2024-04-21T09:43:31.239182Z","shell.execute_reply.started":"2024-04-21T09:43:31.22488Z","shell.execute_reply":"2024-04-21T09:43:31.238078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\ndef build_feature_extractor(model_name='ResNet50'):\n    \"\"\"\n    Builds a feature extraction model using a specified base model from Keras Applications.\n\n    Args:\n        model_name (str): Name of the model to use ('ResNet50', 'VGG16', 'MobileNetV2', etc.).\n\n    Returns:\n        keras.Model: A Keras model that takes an image as input and outputs the extracted features.\n    \"\"\"\n    # Dynamically retrieving the model and preprocessing function based on the model_name\n    base_model_class = getattr(keras.applications, model_name)\n    base_model = base_model_class(\n        weights=\"imagenet\",\n        include_top=False,\n        pooling=\"avg\",\n        input_shape=(IMG_SIZE, IMG_SIZE, 3)\n    )\n    preprocess_input = getattr(keras.applications, model_name.lower()).preprocess_input\n\n    # Define the input layer\n    inputs = keras.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n    # Preprocess input\n    x = preprocess_input(inputs)\n    # Get features\n    outputs = base_model(x)\n\n    # Create the model\n    model = keras.Model(inputs=inputs, outputs=outputs, name=f\"{model_name}_feature_extractor\")\n    \n    return model\n\n# Example usage:\nfeature_extractor = build_feature_extractor('ResNet50')  # I also tried with 'ResNet50', 'VGG16', etc.","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:43:49.967841Z","iopub.execute_input":"2024-04-21T09:43:49.968323Z","iopub.status.idle":"2024-04-21T09:43:56.160463Z","shell.execute_reply.started":"2024-04-21T09:43:49.968245Z","shell.execute_reply":"2024-04-21T09:43:56.15908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport os\n\ndef extract_video_features(dataframe, directory):\n    \"\"\"\n    Processes all videos specified in a dataframe to extract features for model input, including masks.\n\n    Args:\n        dataframe (pd.DataFrame): DataFrame containing indices as video file paths and a 'label' column.\n        directory (str): Base directory where video files are stored.\n\n    Returns:\n        tuple: A tuple (features_and_masks, labels) where features_and_masks contains the video\n               features and masks indicating valid data points, and labels are the binary labels.\n    \"\"\"\n    total_videos = len(dataframe)\n    video_file_paths = dataframe.index.tolist()\n    binary_labels = np.array(dataframe[\"label\"].values == 'FAKE', dtype=int)\n\n    # Initialize arrays to hold data for all videos\n    video_masks = np.zeros((total_videos, MAX_SEQ_LENGTH), dtype=bool)\n    video_features = np.zeros((total_videos, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    # Process each video individually\n    for video_idx, video_file in enumerate(video_file_paths):\n        full_video_path = os.path.join(directory, video_file)\n        video_data = process_video_frames(full_video_path)\n        video_data = np.expand_dims(video_data, axis=0)  # Add a batch dimension\n\n        # Temporary storage for this video's data\n        current_video_mask = np.zeros((1, MAX_SEQ_LENGTH), dtype=bool)\n        current_video_features = np.zeros((1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n        # Frame-by-frame feature extraction\n        frames_to_process = min(MAX_SEQ_LENGTH, video_data.shape[1])\n        for frame_idx in range(frames_to_process):\n            frame = video_data[:, frame_idx, :]\n            extracted_features = feature_extractor.predict(frame[None, :])\n            current_video_features[0, frame_idx, :] = extracted_features\n\n        current_video_mask[0, :frames_to_process] = True  # Mark frames as valid\n\n        # Store the extracted data in the corresponding arrays\n        video_features[video_idx] = current_video_features.squeeze()\n        video_masks[video_idx] = current_video_mask.squeeze()\n\n    return (video_features, video_masks), binary_labels","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:44:14.60535Z","iopub.execute_input":"2024-04-21T09:44:14.60577Z","iopub.status.idle":"2024-04-21T09:44:14.625901Z","shell.execute_reply.started":"2024-04-21T09:44:14.605712Z","shell.execute_reply":"2024-04-21T09:44:14.624455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(train_sample_metadata,test_size=0.1,random_state=42,stratify=train_sample_metadata['label'])\n\nprint(Train_set.shape, Test_set.shape )","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:44:27.563165Z","iopub.execute_input":"2024-04-21T09:44:27.563538Z","iopub.status.idle":"2024-04-21T09:44:28.461825Z","shell.execute_reply.started":"2024-04-21T09:44:27.563481Z","shell.execute_reply":"2024-04-21T09:44:28.459136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, train_labels = extract_video_features(Train_set, \"train\")\ntest_data, test_labels = extract_video_features(Test_set, \"test\")\n\nprint(f\"Frame features in train set: {train_data[0].shape}\")\nprint(f\"Frame masks in train set: {train_data[1].shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:45:01.550148Z","iopub.execute_input":"2024-04-21T09:45:01.550511Z","iopub.status.idle":"2024-04-21T09:45:01.668083Z","shell.execute_reply.started":"2024-04-21T09:45:01.550446Z","shell.execute_reply":"2024-04-21T09:45:01.666892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers, models, regularizers\n\nframe_features_input = layers.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\nmask_input = layers.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n\nx = layers.Bidirectional(layers.GRU(16, return_sequences=True, kernel_regularizer=regularizers.l2(0.01)))(\n    frame_features_input, mask=mask_input\n)\nx = layers.Bidirectional(layers.GRU(8, kernel_regularizer=regularizers.l2(0.01)))(x)\nx = layers.Dropout(0.5)(x)  # Increased dropout\nx = layers.Dense(8, activation=\"relu\")(x)\noutput = layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = models.Model([frame_features_input, mask_input], output)\n\nmodel.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:45:32.768853Z","iopub.execute_input":"2024-04-21T09:45:32.769318Z","iopub.status.idle":"2024-04-21T09:45:36.018338Z","shell.execute_reply.started":"2024-04-21T09:45:32.769246Z","shell.execute_reply":"2024-04-21T09:45:36.017217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom tensorflow.keras import callbacks, models\n\n# Define the directory for storing model checkpoints and the final model\ncheckpoint_dir = './model_checkpoints'\nos.makedirs(checkpoint_dir, exist_ok=True)  # Ensure the directory exists\n\n# Setup the model checkpoint callback to save only the best model during training\ncheckpoint_filepath = os.path.join(checkpoint_dir, 'model-{epoch:02d}-{val_loss:.2f}.h5')\ncheckpoint_callback = callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    monitor='val_loss',\n    verbose=1,\n    save_best_only=True,\n    save_weights_only=True,\n    mode='min'\n)\n\n# EarlyStopping callback to stop training early if no improvement\nearly_stopping_callback = callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    verbose=1,\n    mode='min',\n    restore_best_weights=True\n)\n\n# Model training\nhistory = model.fit(\n    [train_data[0], train_data[1]],\n    train_labels,\n    validation_data=([test_data[0], test_data[1]], test_labels),\n    epochs=10,\n    batch_size=8,\n    callbacks=[checkpoint_callback, early_stopping_callback],\n    verbose=1\n)\n\n# Save the final model after training\nfinal_model_path = os.path.join(checkpoint_dir, 'final_model6.h5')\nmodel.save(final_model_path)\nprint(f\"Model saved to {final_model_path}\")\n\n# Optionally, print the history of training\nprint(\"Training history:\", history.history)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:45:57.2786Z","iopub.execute_input":"2024-04-21T09:45:57.279302Z","iopub.status.idle":"2024-04-21T09:46:29.201331Z","shell.execute_reply.started":"2024-04-21T09:45:57.279145Z","shell.execute_reply":"2024-04-21T09:46:29.199869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test set\ntest_loss, test_accuracy = model.evaluate(\n    [test_data[0], test_data[1]],  # Test features and masks\n    test_labels,                  # Test labels\n    batch_size=8                  # Using the same batch size as during training\n)\n\nprint(f\"Test Loss: {test_loss}\")\nprint(f\"Test Accuracy: {test_accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:48:32.646487Z","iopub.execute_input":"2024-04-21T09:48:32.646932Z","iopub.status.idle":"2024-04-21T09:48:32.754462Z","shell.execute_reply.started":"2024-04-21T09:48:32.646844Z","shell.execute_reply":"2024-04-21T09:48:32.753575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_single_video(frames):\n    frames = frames[None, ...]  # Add batch dimension if not present\n    frame_mask = np.zeros((1, MAX_SEQ_LENGTH), dtype=bool)\n    frame_features = np.zeros((1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype='float32')\n\n    video_length = frames.shape[1]\n    length = min(MAX_SEQ_LENGTH, video_length)\n\n    for j in range(length):\n        frame = np.expand_dims(frames[0, j], axis=0)  # Add batch dimension\n        features = feature_extractor.predict(frame)\n        frame_features[0, j, :] = features\n    \n    frame_mask[0, :length] = True  # Mark the frames that have data\n    return frame_features, frame_mask\n\n\ndef sequence_prediction(video_path):\n    frames = process_video_frames(video_path)\n    frame_features, frame_mask = prepare_single_video(frames)\n    prediction = model.predict([frame_features, frame_mask])[0]\n    return prediction\n\n\nimport imageio\nimport IPython.display as display\n\ndef to_gif(images):\n    imageio.mimsave('animation.gif', images, fps=10)\n    return display.Image(filename='animation.gif')\n\n\ntest_video_path = os.path.join(DATA_FOLDER, TEST_FOLDER, np.random.choice(test_videos[\"video\"].values.tolist()))\nprint(f\"Test video path: {test_video_path}\")\n\nprediction = sequence_prediction(test_video_path)\nif prediction >= 0.5:\n    print('The predicted class of the video is FAKE')\nelse:\n    print('The predicted class of the video is REAL')\n\n# Optional: Load frames again to create a GIF for visualization\nframes = process_video_frames(test_video_path)\nto_gif(frames)","metadata":{"execution":{"iopub.status.busy":"2024-04-21T09:49:00.558591Z","iopub.execute_input":"2024-04-21T09:49:00.559098Z","iopub.status.idle":"2024-04-21T09:49:32.834713Z","shell.execute_reply.started":"2024-04-21T09:49:00.559037Z","shell.execute_reply":"2024-04-21T09:49:32.832538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\n# Compute ROC curve and ROC area\nfpr, tpr, _ = roc_curve(test_labels, test_predictions)\nroc_auc = auc(fpr, tpr)\n\n# Plotting ROC curve\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:07:35.239933Z","iopub.execute_input":"2024-04-21T10:07:35.240313Z","iopub.status.idle":"2024-04-21T10:07:35.545604Z","shell.execute_reply.started":"2024-04-21T10:07:35.240257Z","shell.execute_reply":"2024-04-21T10:07:35.544293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc\n\n# Assuming test_labels and test_predictions are already defined\nfpr, tpr, _ = roc_curve(test_labels, test_predictions)\nroc_auc = auc(fpr, tpr)\n\n# Plotting ROC curve\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\n\n# Save the plot to a file\nplt.savefig('/kaggle/working/ROC_Curve_Plot.png')\nplt.close()\n\n# Provide download instructions\nprint(\"The ROC curve has been saved. You can download it using the link below.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-21T10:10:15.161879Z","iopub.execute_input":"2024-04-21T10:10:15.162369Z","iopub.status.idle":"2024-04-21T10:10:15.361911Z","shell.execute_reply.started":"2024-04-21T10:10:15.162313Z","shell.execute_reply":"2024-04-21T10:10:15.360514Z"},"trusted":true},"execution_count":null,"outputs":[]}]}