{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ENV Setup","metadata":{"_uuid":"d76f2df4-5c9c-4579-a0bd-0791fb963cf2","_cell_guid":"688153e2-d06b-4ac7-9ca3-6b398980f8ab","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"from tensorflow import keras\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"_uuid":"583012ef-f788-44a8-8175-2e7cf83d98a0","_cell_guid":"45b4b703-ec96-4715-9afb-69ab38fa295e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:06.035130Z","iopub.execute_input":"2025-07-17T10:05:06.035498Z","iopub.status.idle":"2025-07-17T10:05:13.942876Z","shell.execute_reply.started":"2025-07-17T10:05:06.035444Z","shell.execute_reply":"2025-07-17T10:05:13.941707Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 64\nEPOCHS = 20\n\nMAX_SEQ_LENGTH = 20\nNUM_FEATURES = 2048","metadata":{"_uuid":"ff84d4e8-d8e5-41ea-a9ad-b4879a83963c","_cell_guid":"6800e631-b6d7-45b0-91ce-719e1620e024","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:04:55.444228Z","iopub.execute_input":"2025-07-17T10:04:55.444530Z","iopub.status.idle":"2025-07-17T10:04:55.449325Z","shell.execute_reply.started":"2025-07-17T10:04:55.444484Z","shell.execute_reply":"2025-07-17T10:04:55.448054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '/kaggle/input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"_uuid":"96b88927-9d23-426f-8707-fa9a02eb401f","_cell_guid":"830d0ac7-a820-442a-ab64-7ce244ec9580","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:15.965440Z","iopub.execute_input":"2025-07-17T10:05:15.965794Z","iopub.status.idle":"2025-07-17T10:05:16.009949Z","shell.execute_reply.started":"2025-07-17T10:05:15.965734Z","shell.execute_reply":"2025-07-17T10:05:16.008883Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{"_uuid":"c71c3cfc-7c92-48c4-88f2-2d903adc18cc","_cell_guid":"f733c96d-76b6-48ef-b14f-8aaf5ce23923","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"_uuid":"5a726c60-ca8f-46c2-b3f4-aecddd510419","_cell_guid":"78ed1674-d8ee-47b1-95f8-c0399de7192a","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:17.655362Z","iopub.execute_input":"2025-07-17T10:05:17.655728Z","iopub.status.idle":"2025-07-17T10:05:18.023560Z","shell.execute_reply.started":"2025-07-17T10:05:17.655665Z","shell.execute_reply":"2025-07-17T10:05:18.022703Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()","metadata":{"_uuid":"d36ab2eb-d98e-46c8-97c4-cd02f79f65dd","_cell_guid":"abea6c55-093e-4ce4-b64d-a7f90686c2fc","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:19.960285Z","iopub.execute_input":"2025-07-17T10:05:19.960644Z","iopub.status.idle":"2025-07-17T10:05:20.241045Z","shell.execute_reply.started":"2025-07-17T10:05:19.960579Z","shell.execute_reply":"2025-07-17T10:05:20.239529Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.shape","metadata":{"_uuid":"3660873d-c6ec-4c90-bcc4-87571dcc28d0","_cell_guid":"e9d96dc6-120c-462f-aa8d-272ac698dcd5","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:23.475491Z","iopub.execute_input":"2025-07-17T10:05:23.475802Z","iopub.status.idle":"2025-07-17T10:05:23.482124Z","shell.execute_reply.started":"2025-07-17T10:05:23.475756Z","shell.execute_reply":"2025-07-17T10:05:23.481253Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(10).index)\nfake_train_sample_video","metadata":{"_uuid":"a96675ed-064e-4685-9509-70107f95853e","_cell_guid":"dce7a62d-4baf-445d-b686-40fd4484bb64","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:23.810385Z","iopub.execute_input":"2025-07-17T10:05:23.810714Z","iopub.status.idle":"2025-07-17T10:05:23.820847Z","shell.execute_reply.started":"2025-07-17T10:05:23.810665Z","shell.execute_reply":"2025-07-17T10:05:23.819822Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef show_first_frame(video_file_path):\n    \n    video_capture = cv2.VideoCapture(video_file_path)\n    \n    # Verify that the video file was opened successfully\n    if not video_capture.isOpened():\n        video_capture.release()  # Ensure resources are released\n        raise FileNotFoundError(f\"Failed to access the video at {video_file_path}\")\n    \n    # Attempt to capture the first frame\n    successful, frame = video_capture.read()\n    \n    # Ensure that a frame was successfully captured\n    if not successful:\n        video_capture.release()  # Ensure resources are released before raising an error\n        raise RuntimeError(\"No frames could be read from the video file\")\n    \n    # Adjust the frame's color format for displaying\n    frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n    \n    # Display the frame using matplotlib\n    plt.figure(figsize=(10, 10))\n    plt.imshow(frame_rgb)\n    plt.axis('off')  # Hide axes for better visualization\n    plt.show()\n    \n    # Close the video capture object to free resources\n    video_capture.release()","metadata":{"_uuid":"cbf75736-5e92-4e85-ac81-370bdf82dbab","_cell_guid":"5ec42512-fd17-4d32-85c6-b299f6da587d","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:24.415740Z","iopub.execute_input":"2025-07-17T10:05:24.416127Z","iopub.status.idle":"2025-07-17T10:05:24.424159Z","shell.execute_reply.started":"2025-07-17T10:05:24.416058Z","shell.execute_reply":"2025-07-17T10:05:24.422873Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    show_first_frame(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"_uuid":"9618cd25-ad9f-45c3-9a91-13526681d2a9","_cell_guid":"4055bed4-e9fe-4599-9281-bd321762e92c","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:28.005173Z","iopub.execute_input":"2025-07-17T10:05:28.005507Z","iopub.status.idle":"2025-07-17T10:05:32.015117Z","shell.execute_reply.started":"2025-07-17T10:05:28.005458Z","shell.execute_reply":"2025-07-17T10:05:32.014234Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(10).index)\nreal_train_sample_video","metadata":{"_uuid":"06e1eabf-8d5f-4a19-9e8f-a17067952c91","_cell_guid":"3f64a41e-18e5-4cb9-9f75-bae72a2c692c","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:32.017844Z","iopub.execute_input":"2025-07-17T10:05:32.018160Z","iopub.status.idle":"2025-07-17T10:05:32.026854Z","shell.execute_reply.started":"2025-07-17T10:05:32.018105Z","shell.execute_reply":"2025-07-17T10:05:32.025648Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    show_first_frame(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"_uuid":"5c2dec07-7f0d-4f6b-a646-f47ad97a9897","_cell_guid":"b445379b-c702-40be-9fe9-810615bd3c7f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:32.028556Z","iopub.execute_input":"2025-07-17T10:05:32.028822Z","iopub.status.idle":"2025-07-17T10:05:36.086054Z","shell.execute_reply.started":"2025-07-17T10:05:32.028780Z","shell.execute_reply":"2025-07-17T10:05:36.085090Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata['original'].value_counts()[0:10]","metadata":{"_uuid":"6e95ecf0-c779-43ff-b581-1c5bcfc07425","_cell_guid":"577eaf89-f2ef-405e-9099-4f0c3f6b8572","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:40.120169Z","iopub.execute_input":"2025-07-17T10:05:40.120590Z","iopub.status.idle":"2025-07-17T10:05:40.130762Z","shell.execute_reply.started":"2025-07-17T10:05:40.120526Z","shell.execute_reply":"2025-07-17T10:05:40.129310Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\n\ndef show_frames_from_videos(video_files, base_folder=TRAIN_SAMPLE_FOLDER):\n    \n    fig, axes = plt.subplots(2, 3, figsize=(16, 8))\n    \n    # Loop through the first six videos in the list\n    for index, video_name in enumerate(video_files[:6]):\n        video_full_path = os.path.join(DATA_FOLDER, base_folder, video_name)\n        video_capture = cv2.VideoCapture(video_full_path)\n        \n        success, frame = video_capture.read()\n        if not success:\n            print(f\"Failed to read from {video_name}\")\n            axes[index // 3, index % 3].set_title(\"Failed to load video\")\n            axes[index // 3, index % 3].axis('off')\n            continue\n        \n        # Convert the color from BGR to RGB\n        frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        \n        # Display the image in the respective subplot\n        axes[index // 3, index % 3].imshow(frame_rgb)\n        axes[index // 3, index % 3].set_title(video_name)\n        axes[index // 3, index % 3].axis('on')  # Keep the axis on for clarity\n        \n        # Release the video capture object\n        video_capture.release()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"_uuid":"c2cdc3b1-2f42-469e-9368-c922e0068846","_cell_guid":"b577d781-0b50-4348-981d-0420a4458688","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:40.331667Z","iopub.execute_input":"2025-07-17T10:05:40.332030Z","iopub.status.idle":"2025-07-17T10:05:40.343071Z","shell.execute_reply.started":"2025-07-17T10:05:40.331960Z","shell.execute_reply":"2025-07-17T10:05:40.341861Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='meawmsgiti.mp4'].index)\nshow_frames_from_videos(same_original_fake_train_sample_video)","metadata":{"_uuid":"02f1281d-97e1-49b2-90a5-03edf189fd19","_cell_guid":"9a5069df-81b4-421d-8c14-747debfadb6f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:42.896694Z","iopub.execute_input":"2025-07-17T10:05:42.897030Z","iopub.status.idle":"2025-07-17T10:05:44.965821Z","shell.execute_reply.started":"2025-07-17T10:05:42.896978Z","shell.execute_reply":"2025-07-17T10:05:44.964631Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"_uuid":"8686565a-88df-4015-aaa6-b45a28ffa3dd","_cell_guid":"2474aae7-9404-4f70-ba3f-173eac25ee81","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:44.968186Z","iopub.execute_input":"2025-07-17T10:05:44.968656Z","iopub.status.idle":"2025-07-17T10:05:44.976854Z","shell.execute_reply.started":"2025-07-17T10:05:44.968573Z","shell.execute_reply":"2025-07-17T10:05:44.975402Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_videos.head(10)","metadata":{"_uuid":"dd20fe9a-c0f4-4acb-8fbc-37ab47042ecc","_cell_guid":"8c1d7beb-14c9-43b6-9911-10329c0c5d69","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:44.978804Z","iopub.execute_input":"2025-07-17T10:05:44.979326Z","iopub.status.idle":"2025-07-17T10:05:45.001621Z","shell.execute_reply.started":"2025-07-17T10:05:44.979223Z","shell.execute_reply":"2025-07-17T10:05:45.000431Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_first_frame(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[3].video))","metadata":{"_uuid":"648fab43-952c-433d-b1f2-8c38d3d6901e","_cell_guid":"73d8613f-3b58-4edb-854b-abd5dfad5527","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:45.003855Z","iopub.execute_input":"2025-07-17T10:05:45.004611Z","iopub.status.idle":"2025-07-17T10:05:45.377540Z","shell.execute_reply.started":"2025-07-17T10:05:45.004380Z","shell.execute_reply":"2025-07-17T10:05:45.376361Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fake_videos = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].index)","metadata":{"_uuid":"4e61542b-65a5-404a-baf4-b9d67eb3252d","_cell_guid":"772238df-7618-488b-9474-8d18776dbfa4","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:45.380409Z","iopub.execute_input":"2025-07-17T10:05:45.380856Z","iopub.status.idle":"2025-07-17T10:05:45.388888Z","shell.execute_reply.started":"2025-07-17T10:05:45.380746Z","shell.execute_reply":"2025-07-17T10:05:45.387782Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\nimport os\n\ndef embed_video_in_notebook(video_filename, directory=TRAIN_SAMPLE_FOLDER):\n    try:\n        # Construct the full path to the video file\n        video_path = os.path.join(DATA_FOLDER, directory, video_filename)\n        \n        # Read the video file as binary data\n        with open(video_path, 'rb') as video_file:\n            video_data = video_file.read()\n\n        # Encode the video data in base64 and create the data URL\n        video_base64 = b64encode(video_data).decode('utf-8')\n        data_url = f\"data:video/mp4;base64,{video_base64}\"\n\n        # Return an HTML object that contains the video element\n        return HTML(f'<video width=\"500\" controls><source src=\"{data_url}\" type=\"video/mp4\"></video>')\n    \n    except FileNotFoundError:\n        raise FileNotFoundError(f\"The video file {video_filename} could not be found in {directory}.\")\n\n# Example usage:\n# Assuming 'fake_videos[10]' contains the filename of the video to play\nvideo_to_play = fake_videos[14] \nembed_video_in_notebook(video_to_play)","metadata":{"_uuid":"f40d83fd-1b14-40b5-ab7f-07ca9ce76c3f","_cell_guid":"8533ad9b-5121-4eb5-b74d-2acd56ea5c6e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:47.485665Z","iopub.execute_input":"2025-07-17T10:05:47.485978Z","iopub.status.idle":"2025-07-17T10:05:47.629126Z","shell.execute_reply.started":"2025-07-17T10:05:47.485920Z","shell.execute_reply":"2025-07-17T10:05:47.627632Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef square_crop_frame(image):\n    \n    height, width = image.shape[:2]\n    min_dimension = min(height, width)\n    start_x = (width - min_dimension) // 2\n    start_y = (height - min_dimension) // 2\n    return image[start_y:start_y + min_dimension, start_x:start_x + min_dimension]\n\ndef process_video_frames(video_path, max_frames=0, resize_dims=(IMG_SIZE, IMG_SIZE)):\n\n    capture = cv2.VideoCapture(video_path)\n    processed_frames = []\n    try:\n        while True:\n            read_success, frame = capture.read()\n            if not read_success:\n                break\n            frame = square_crop_frame(frame)\n            frame = cv2.resize(frame, resize_dims)\n            # Convert BGR to RGB for standard color format\n            frame = frame[..., ::-1]\n            processed_frames.append(frame)\n\n            if max_frames > 0 and len(processed_frames) >= max_frames:\n                break\n    finally:\n        capture.release()\n    return np.array(processed_frames)","metadata":{"_uuid":"2485388e-2653-4c0f-9ec7-c9292179e031","_cell_guid":"8d690f8a-2f35-41d0-9a9f-9228844f0e8b","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:47.752891Z","iopub.execute_input":"2025-07-17T10:05:47.753418Z","iopub.status.idle":"2025-07-17T10:05:47.763146Z","shell.execute_reply.started":"2025-07-17T10:05:47.753163Z","shell.execute_reply":"2025-07-17T10:05:47.761901Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\ndef build_feature_extractor(model_name='ResNet50'):\n    # Get model class and preprocessing function\n    base_model_class = getattr(keras.applications, model_name)\n    preprocess_input = getattr(keras.applications, model_name.lower()).preprocess_input\n\n    # Base model without top layers\n    base_model = base_model_class(\n        weights='imagenet',\n        include_top=False,\n        pooling='avg',\n        input_shape=(IMG_SIZE, IMG_SIZE, 3)\n    )\n\n    # Build sequential model\n    model = keras.Sequential([\n        keras.layers.Lambda(preprocess_input, input_shape=(IMG_SIZE, IMG_SIZE, 3), name='preprocessing'),\n        base_model\n    ], name=f\"{model_name}_feature_extractor_seq\")\n\n    return model","metadata":{"_uuid":"ca8aa0c0-3f0e-467f-8a30-a0f57da34980","_cell_guid":"233eeff0-db54-4d43-b5ef-b4266f321eb7","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:05:50.855083Z","iopub.execute_input":"2025-07-17T10:05:50.855427Z","iopub.status.idle":"2025-07-17T10:05:50.862550Z","shell.execute_reply.started":"2025-07-17T10:05:50.855375Z","shell.execute_reply":"2025-07-17T10:05:50.861431Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training","metadata":{"_uuid":"233573b8-dbed-4b68-9cfc-c946fadb78b4","_cell_guid":"7f7d2b29-d04f-42cd-95d1-79d04f300070","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(train_sample_metadata,test_size=0.1,random_state=42,stratify=train_sample_metadata['label'])\n\nprint(Train_set.shape, Test_set.shape )","metadata":{"_uuid":"5e0706f6-2ba2-4452-bd70-031e10cbbabd","_cell_guid":"359749a3-db5a-461e-bcba-b49b0cdb1a23","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:06:14.245716Z","iopub.execute_input":"2025-07-17T10:06:14.246050Z","iopub.status.idle":"2025-07-17T10:06:15.266296Z","shell.execute_reply.started":"2025-07-17T10:06:14.245998Z","shell.execute_reply":"2025-07-17T10:06:15.265378Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN(InceptionV3) + LSTM","metadata":{"_uuid":"ca77c841-de63-44d1-9cd6-a8bf42a835e0","_cell_guid":"3c99efa5-58b6-4731-b14f-8c9883b56429","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import numpy as np\nimport os\n\nfeature_extractor = build_feature_extractor('ResNet50')\n\ndef extract_video_features(dataframe, directory):\n    \n    total_videos = len(dataframe)\n    video_file_paths = dataframe.index.tolist()\n    binary_labels = np.array(dataframe[\"label\"].values == 'FAKE', dtype=int)\n\n    # Initialize arrays to hold data for all videos\n    video_masks = np.zeros((total_videos, MAX_SEQ_LENGTH), dtype=bool)\n    video_features = np.zeros((total_videos, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    # Process each video individually\n    for video_idx, video_file in enumerate(video_file_paths):\n        full_video_path = os.path.join(directory, video_file)\n        video_data = process_video_frames(full_video_path)\n        video_data = np.expand_dims(video_data, axis=0)  # Add a batch dimension\n\n        # Temporary storage for this video's data\n        current_video_mask = np.zeros((1, MAX_SEQ_LENGTH), dtype=bool)\n        current_video_features = np.zeros((1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n        # Frame-by-frame feature extraction\n        frames_to_process = min(MAX_SEQ_LENGTH, video_data.shape[1])\n        for frame_idx in range(frames_to_process):\n            frame = video_data[:, frame_idx, :]\n            extracted_features = feature_extractor.predict(frame[None, :])\n            current_video_features[0, frame_idx, :] = extracted_features\n\n        current_video_mask[0, :frames_to_process] = True  # Mark frames as valid\n\n        # Store the extracted data in the corresponding arrays\n        video_features[video_idx] = current_video_features.squeeze()\n        video_masks[video_idx] = current_video_mask.squeeze()\n\n    return (video_features, video_masks), binary_labels","metadata":{"_uuid":"fd44b407-7234-49c5-8dcf-4b60796f7f0d","_cell_guid":"a9460274-f9aa-4567-bfc5-fcdd88bd7b78","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:10:08.374957Z","iopub.execute_input":"2025-07-17T10:10:08.375323Z","iopub.status.idle":"2025-07-17T10:10:17.735439Z","shell.execute_reply.started":"2025-07-17T10:10:08.375257Z","shell.execute_reply":"2025-07-17T10:10:17.734114Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data, train_labels = extract_video_features(Train_set, \"train\")\ntest_data, test_labels = extract_video_features(Test_set, \"test\")\n\nprint(f\"Frame features in train set: {train_data[0].shape}\")\nprint(f\"Frame masks in train set: {train_data[1].shape}\")","metadata":{"_uuid":"b884511d-2f83-4aa8-9507-e7df2e435b99","_cell_guid":"b45c55b5-f216-499a-b2ba-56415fc3f619","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:10:24.587029Z","iopub.execute_input":"2025-07-17T10:10:24.587420Z","iopub.status.idle":"2025-07-17T10:10:24.688289Z","shell.execute_reply.started":"2025-07-17T10:10:24.587359Z","shell.execute_reply":"2025-07-17T10:10:24.687308Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models, regularizers, metrics\n\nframe_features_input = layers.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\nmask_input = layers.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n\nx = layers.LSTM(\n    16, return_sequences=True, kernel_regularizer=regularizers.l2(0.01)\n)(frame_features_input, mask=mask_input)\n\nx = layers.LSTM(\n    8, kernel_regularizer=regularizers.l2(0.01)\n)(x)\n\nx = layers.Dropout(0.5)(x)\nx = layers.Dense(8, activation=\"relu\")(x)\noutput = layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = models.Model([frame_features_input, mask_input], output)\n\nmodel.compile(\n    loss=\"binary_crossentropy\",\n    optimizer=\"adam\",\n    metrics=[\n        \"accuracy\",\n        metrics.Recall(name=\"recall\"),\n        metrics.Precision(name=\"precision\")\n    ]\n)\n\nmodel.summary()","metadata":{"_uuid":"69c2d8ea-39ac-4420-8d75-e71f96cdd1ab","_cell_guid":"c664a447-f4a8-4fe6-86a6-333fdeb1ef61","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:10:24.969843Z","iopub.execute_input":"2025-07-17T10:10:24.970156Z","iopub.status.idle":"2025-07-17T10:10:26.547219Z","shell.execute_reply.started":"2025-07-17T10:10:24.970105Z","shell.execute_reply":"2025-07-17T10:10:26.546314Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tensorflow.keras import callbacks, models\n\n# Define the directory for storing model checkpoints and the final model\ncheckpoint_dir = './model_checkpoints'\nos.makedirs(checkpoint_dir, exist_ok=True)  # Ensure the directory exists\n\n# Setup the model checkpoint callback to save only the best model during training\ncheckpoint_filepath = os.path.join(checkpoint_dir, 'model-{epoch:02d}-{val_loss:.2f}.h5')\ncheckpoint_callback = callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    monitor='val_loss',\n    verbose=1,\n    save_best_only=True,\n    save_weights_only=True,\n    mode='min'\n)\n\n# EarlyStopping callback to stop training early if no improvement\nearly_stopping_callback = callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    verbose=1,\n    mode='min',\n    restore_best_weights=True\n)\n\n# Model training\nhistory = model.fit(\n    [train_data[0], train_data[1]],\n    train_labels,\n    validation_data=([test_data[0], test_data[1]], test_labels),\n    epochs=10,\n    batch_size=8,\n    callbacks=[checkpoint_callback, early_stopping_callback],\n    verbose=1\n)\n\n# Save the final model after training\nfinal_model_path = os.path.join(checkpoint_dir, 'final_model6.h5')\nmodel.save(final_model_path)\nprint(f\"Model saved to {final_model_path}\")\n\n# Optionally, print the history of training\nprint(\"Training history:\", history.history)","metadata":{"_uuid":"ebb5d677-aa52-4036-aaef-46d7a56152ff","_cell_guid":"42de1018-d04e-4ace-ac59-93701c77dcfa","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:10:31.249088Z","iopub.execute_input":"2025-07-17T10:10:31.249478Z","iopub.status.idle":"2025-07-17T10:10:49.576889Z","shell.execute_reply.started":"2025-07-17T10:10:31.249424Z","shell.execute_reply":"2025-07-17T10:10:49.575802Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Model Evolution","metadata":{"_uuid":"9191b865-5d87-42b5-af9a-5bde452609cf","_cell_guid":"171b9994-405d-400c-a7d5-97fe89ea02d2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Evaluate the model on the test set\ntest_loss, test_accuracy, test_recall, test_precision = model.evaluate(\n    [test_data[0], test_data[1]],  # Test features and masks\n    test_labels,                   # Test labels\n    batch_size=8                  # Use the batch size consistent with training\n)\n\nprint(f\"Test Loss: {test_loss:.4f}\")\nprint(f\"Test Accuracy: {test_accuracy:.4f}\")\nprint(f\"Test Recall: {test_recall:.4f}\")\nprint(f\"Test Precision: {test_precision:.4f}\")","metadata":{"_uuid":"0d0763fa-4779-4a21-ae5f-1e37b2a7c2d2","_cell_guid":"90901546-6d29-478b-8b37-0b6e532bcb7c","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:10:55.142440Z","iopub.execute_input":"2025-07-17T10:10:55.142783Z","iopub.status.idle":"2025-07-17T10:10:55.221180Z","shell.execute_reply.started":"2025-07-17T10:10:55.142719Z","shell.execute_reply":"2025-07-17T10:10:55.220365Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Assuming `history` is the result of model.fit()\nhist = history.history\n\n# Create 2x2 subplot\nplt.figure(figsize=(14, 10))\n\n# Plot Accuracy\nplt.subplot(2, 2, 1)\nplt.plot(hist['accuracy'], label='Train Accuracy')\nplt.plot(hist['val_accuracy'], label='Val Accuracy')\nplt.title('Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# Plot Loss\nplt.subplot(2, 2, 2)\nplt.plot(hist['loss'], label='Train Loss')\nplt.plot(hist['val_loss'], label='Val Loss')\nplt.title('Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\n# Plot Recall\nplt.subplot(2, 2, 3)\nplt.plot(hist['recall'], label='Train Recall')\nplt.plot(hist['val_recall'], label='Val Recall')\nplt.title('Recall')\nplt.xlabel('Epoch')\nplt.ylabel('Recall')\nplt.legend()\n\n# Plot Precision\nplt.subplot(2, 2, 4)\nplt.plot(hist['precision'], label='Train Precision')\nplt.plot(hist['val_precision'], label='Val Precision')\nplt.title('Precision')\nplt.xlabel('Epoch')\nplt.ylabel('Precision')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"37168689-55c7-48d6-a1cf-59d88d7aa763","_cell_guid":"ef4ea449-bd7d-4489-b180-c2fa821a10ff","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:10:55.506143Z","iopub.execute_input":"2025-07-17T10:10:55.506559Z","iopub.status.idle":"2025-07-17T10:10:56.578131Z","shell.execute_reply.started":"2025-07-17T10:10:55.506492Z","shell.execute_reply":"2025-07-17T10:10:56.577162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_prob = model.predict([test_data[0], test_data[1]], batch_size=8)\ny_pred = (y_pred_prob > 0.5).astype(int).flatten()","metadata":{"_uuid":"98400e13-66a2-4dce-b885-dce38968f6f4","_cell_guid":"e2a9d77a-db09-4ebc-a330-c80f8d102c34","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:10:56.580122Z","iopub.execute_input":"2025-07-17T10:10:56.580444Z","iopub.status.idle":"2025-07-17T10:10:58.141183Z","shell.execute_reply.started":"2025-07-17T10:10:56.580396Z","shell.execute_reply":"2025-07-17T10:10:58.140318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Classification report\nprint(classification_report(test_labels, y_pred, digits=4))","metadata":{"_uuid":"e3f1363a-d804-433c-8ab1-8b2911ce64eb","_cell_guid":"510577c3-249f-4ef2-88e6-5ecac5a72d2d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:11:24.360178Z","iopub.execute_input":"2025-07-17T10:11:24.360548Z","iopub.status.idle":"2025-07-17T10:11:24.499579Z","shell.execute_reply.started":"2025-07-17T10:11:24.360494Z","shell.execute_reply":"2025-07-17T10:11:24.498490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion matrix\ncm = confusion_matrix(test_labels, y_pred)\n\n# Plot confusion matrix\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"_uuid":"bdfe78de-ea40-426c-9517-e46f69295ff1","_cell_guid":"3aa9a36f-4e70-41dc-aa2f-208b38de29ba","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:11:31.203996Z","iopub.execute_input":"2025-07-17T10:11:31.204345Z","iopub.status.idle":"2025-07-17T10:11:31.920110Z","shell.execute_reply.started":"2025-07-17T10:11:31.204287Z","shell.execute_reply":"2025-07-17T10:11:31.917792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN(VGG16) + Bidirectional GRU","metadata":{"_uuid":"b0a9678e-5b36-437d-9d96-50a870e0cfbf","_cell_guid":"95c8a4cd-82c1-4fb8-9842-47fd182c7856","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import numpy as np\nimport os\n\nfeature_extractor = build_feature_extractor('VGG16')\n\ndef extract_video_features(dataframe, directory):\n    \n    total_videos = len(dataframe)\n    video_file_paths = dataframe.index.tolist()\n    binary_labels = np.array(dataframe[\"label\"].values == 'FAKE', dtype=int)\n\n    # Initialize arrays to hold data for all videos\n    video_masks = np.zeros((total_videos, MAX_SEQ_LENGTH), dtype=bool)\n    video_features = np.zeros((total_videos, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n    # Process each video individually\n    for video_idx, video_file in enumerate(video_file_paths):\n        full_video_path = os.path.join(directory, video_file)\n        video_data = process_video_frames(full_video_path)\n        video_data = np.expand_dims(video_data, axis=0)  # Add a batch dimension\n\n        # Temporary storage for this video's data\n        current_video_mask = np.zeros((1, MAX_SEQ_LENGTH), dtype=bool)\n        current_video_features = np.zeros((1, MAX_SEQ_LENGTH, NUM_FEATURES), dtype=\"float32\")\n\n        # Frame-by-frame feature extraction\n        frames_to_process = min(MAX_SEQ_LENGTH, video_data.shape[1])\n        for frame_idx in range(frames_to_process):\n            frame = video_data[:, frame_idx, :]\n            extracted_features = feature_extractor.predict(frame[None, :])\n            current_video_features[0, frame_idx, :] = extracted_features\n\n        current_video_mask[0, :frames_to_process] = True  # Mark frames as valid\n\n        # Store the extracted data in the corresponding arrays\n        video_features[video_idx] = current_video_features.squeeze()\n        video_masks[video_idx] = current_video_mask.squeeze()\n\n    return (video_features, video_masks), binary_labels","metadata":{"_uuid":"8e02e35b-e5c5-405c-bd09-98348ce2733c","_cell_guid":"cbb653ff-3ca0-461f-bc39-bc002015bc0a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:11:36.067446Z","iopub.execute_input":"2025-07-17T10:11:36.067844Z","iopub.status.idle":"2025-07-17T10:11:38.933518Z","shell.execute_reply.started":"2025-07-17T10:11:36.067789Z","shell.execute_reply":"2025-07-17T10:11:38.932340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models, regularizers, metrics\n\nframe_features_input = layers.Input((MAX_SEQ_LENGTH, NUM_FEATURES))\nmask_input = layers.Input((MAX_SEQ_LENGTH,), dtype=\"bool\")\n\nx = layers.Bidirectional(layers.LSTM(16, return_sequences=True, kernel_regularizer=regularizers.l2(0.01)))(\n    frame_features_input, mask=mask_input\n)\nx = layers.Bidirectional(layers.LSTM(8, kernel_regularizer=regularizers.l2(0.01)))(x)\nx = layers.Dropout(0.5)(x)\nx = layers.Dense(8, activation=\"relu\")(x)\noutput = layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = models.Model([frame_features_input, mask_input], output)\n\nmodel.compile(\n    loss=\"binary_crossentropy\",\n    optimizer=\"adam\",\n    metrics=[\n        \"accuracy\",\n        metrics.Recall(name=\"recall\"),\n        metrics.Precision(name=\"precision\")\n    ]\n)\n\nmodel.summary()","metadata":{"_uuid":"88e26d1b-af32-4753-9537-8314059d5918","_cell_guid":"d80f9c89-b710-45c3-a8f4-7d7f4199d1df","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:11:45.828031Z","iopub.execute_input":"2025-07-17T10:11:45.828368Z","iopub.status.idle":"2025-07-17T10:11:48.730264Z","shell.execute_reply.started":"2025-07-17T10:11:45.828316Z","shell.execute_reply":"2025-07-17T10:11:48.729263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tensorflow.keras import callbacks, models\n\n# Define the directory for storing model checkpoints and the final model\ncheckpoint_dir = './model_checkpoints'\nos.makedirs(checkpoint_dir, exist_ok=True)  # Ensure the directory exists\n\n# Setup the model checkpoint callback to save only the best model during training\ncheckpoint_filepath = os.path.join(checkpoint_dir, 'bid-model-{epoch:02d}-{val_loss:.2f}.h5')\ncheckpoint_callback = callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    monitor='val_loss',\n    verbose=1,\n    save_best_only=True,\n    save_weights_only=True,\n    mode='min'\n)\n\n# EarlyStopping callback to stop training early if no improvement\nearly_stopping_callback = callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=10,\n    verbose=1,\n    mode='min',\n    restore_best_weights=True\n)\n\n# Model training\nhistory = model.fit(\n    [train_data[0], train_data[1]],\n    train_labels,\n    validation_data=([test_data[0], test_data[1]], test_labels),\n    epochs=10,\n    batch_size=8,\n    callbacks=[checkpoint_callback, early_stopping_callback],\n    verbose=1\n)\n\n# Save the final model after training\nfinal_model_path = os.path.join(checkpoint_dir, 'final_model6.h5')\nmodel.save(final_model_path)\nprint(f\"Model saved to {final_model_path}\")\n\n# Optionally, print the history of training\nprint(\"Training history:\", history.history)","metadata":{"_uuid":"114d7a73-1ba3-41cc-ad5c-a2134b994968","_cell_guid":"d7e69c28-6d57-4ef9-afc8-11b1c7970134","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:11:48.732606Z","iopub.execute_input":"2025-07-17T10:11:48.732868Z","iopub.status.idle":"2025-07-17T10:12:16.592752Z","shell.execute_reply.started":"2025-07-17T10:11:48.732824Z","shell.execute_reply":"2025-07-17T10:12:16.591328Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Model Evolution","metadata":{"_uuid":"972a7e50-35a9-40fc-957f-503be7bc37d0","_cell_guid":"8faadcbc-462a-4674-acdc-915a3e860ea8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Evaluate the model on the test set\ntest_loss, test_accuracy, test_recall, test_precision = model.evaluate(\n    [test_data[0], test_data[1]],  # Test features and masks\n    test_labels,                   # Test labels\n    batch_size=8                  # Use the batch size consistent with training\n)\n\nprint(f\"Test Loss: {test_loss:.4f}\")\nprint(f\"Test Accuracy: {test_accuracy:.4f}\")\nprint(f\"Test Recall: {test_recall:.4f}\")\nprint(f\"Test Precision: {test_precision:.4f}\")","metadata":{"_uuid":"05987ec2-0462-4287-a291-49a6ac4e0473","_cell_guid":"a6d564a0-4da6-4e4c-81f0-69da2a0fdd0b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:12:16.594166Z","iopub.execute_input":"2025-07-17T10:12:16.594458Z","iopub.status.idle":"2025-07-17T10:12:16.679403Z","shell.execute_reply.started":"2025-07-17T10:12:16.594413Z","shell.execute_reply":"2025-07-17T10:12:16.677915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Assuming `history` is the result of model.fit()\nhist = history.history\n\n# Create 2x2 subplot\nplt.figure(figsize=(14, 10))\n\n# Plot Accuracy\nplt.subplot(2, 2, 1)\nplt.plot(hist['accuracy'], label='Train Accuracy')\nplt.plot(hist['val_accuracy'], label='Val Accuracy')\nplt.title('Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\n# Plot Loss\nplt.subplot(2, 2, 2)\nplt.plot(hist['loss'], label='Train Loss')\nplt.plot(hist['val_loss'], label='Val Loss')\nplt.title('Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\n# Plot Recall\nplt.subplot(2, 2, 3)\nplt.plot(hist['recall'], label='Train Recall')\nplt.plot(hist['val_recall'], label='Val Recall')\nplt.title('Recall')\nplt.xlabel('Epoch')\nplt.ylabel('Recall')\nplt.legend()\n\n# Plot Precision\nplt.subplot(2, 2, 4)\nplt.plot(hist['precision'], label='Train Precision')\nplt.plot(hist['val_precision'], label='Val Precision')\nplt.title('Precision')\nplt.xlabel('Epoch')\nplt.ylabel('Precision')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"84c0b474-5b34-4104-9aa4-f3ffd878cb21","_cell_guid":"f6006003-83f3-4be9-9b28-4bc84395f0e7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:12:16.681293Z","iopub.execute_input":"2025-07-17T10:12:16.681616Z","iopub.status.idle":"2025-07-17T10:12:18.341772Z","shell.execute_reply.started":"2025-07-17T10:12:16.681559Z","shell.execute_reply":"2025-07-17T10:12:18.340752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_prob = model.predict([test_data[0], test_data[1]], batch_size=8)\ny_pred = (y_pred_prob > 0.5).astype(int).flatten()","metadata":{"_uuid":"c502623e-f899-4a35-a302-d0d23c9be3b7","_cell_guid":"9a4c1aef-ca10-48ce-b6f7-cd6467f32198","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:12:18.343890Z","iopub.execute_input":"2025-07-17T10:12:18.344263Z","iopub.status.idle":"2025-07-17T10:12:21.184364Z","shell.execute_reply.started":"2025-07-17T10:12:18.344212Z","shell.execute_reply":"2025-07-17T10:12:21.183482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Classification report\nprint(classification_report(test_labels, y_pred, digits=4))","metadata":{"_uuid":"c7e4c64f-476c-4949-a215-eb2480323cd2","_cell_guid":"9b242ab7-e7d1-4086-beff-f708d96a462b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:12:44.676084Z","iopub.execute_input":"2025-07-17T10:12:44.676460Z","iopub.status.idle":"2025-07-17T10:12:44.688674Z","shell.execute_reply.started":"2025-07-17T10:12:44.676405Z","shell.execute_reply":"2025-07-17T10:12:44.687769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion matrix\ncm = confusion_matrix(test_labels, y_pred)\n\n# Plot confusion matrix\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"_uuid":"087a5f73-f1dd-452d-b48b-f1f9ff06e9ce","_cell_guid":"b005dc88-c4de-4c21-953a-7752cdfca875","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-17T10:12:44.927975Z","iopub.execute_input":"2025-07-17T10:12:44.928317Z","iopub.status.idle":"2025-07-17T10:12:45.179353Z","shell.execute_reply.started":"2025-07-17T10:12:44.928267Z","shell.execute_reply":"2025-07-17T10:12:45.178074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\n# Compute ROC curve and ROC area\nfpr, tpr, _ = roc_curve(test_labels, y_pred)\nroc_auc = auc(fpr, tpr)\n\n# Plotting ROC curve\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"_uuid":"69c64f2f-cc1c-4047-a8b6-8b13b074ad6d","_cell_guid":"83588d9a-d830-4226-aafd-51e2902ccd77","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-17T10:13:42.954926Z","iopub.execute_input":"2025-07-17T10:13:42.955307Z","iopub.status.idle":"2025-07-17T10:13:43.203361Z","shell.execute_reply.started":"2025-07-17T10:13:42.955241Z","shell.execute_reply":"2025-07-17T10:13:43.202053Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}