{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":18147,"sourceType":"datasetVersion","datasetId":13405},{"sourceId":842050,"sourceType":"datasetVersion","datasetId":444558},{"sourceId":893807,"sourceType":"datasetVersion","datasetId":451078},{"sourceId":6358196,"sourceType":"datasetVersion","datasetId":3579787}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-13T06:48:27.396887Z","iopub.execute_input":"2024-06-13T06:48:27.397192Z","iopub.status.idle":"2024-06-13T06:48:27.900327Z","shell.execute_reply.started":"2024-06-13T06:48:27.397149Z","shell.execute_reply":"2024-06-13T06:48:27.899401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install numpy","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:56:34.927296Z","iopub.execute_input":"2024-06-13T05:56:34.927634Z","iopub.status.idle":"2024-06-13T05:56:40.842544Z","shell.execute_reply.started":"2024-06-13T05:56:34.927588Z","shell.execute_reply":"2024-06-13T05:56:40.841513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall torch torchvision -y\n!pip install torch==1.7.1 torchvision==0.8.2\n!pip install facenet-pytorch\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:58:06.571068Z","iopub.execute_input":"2024-06-13T05:58:06.571392Z","iopub.status.idle":"2024-06-13T05:59:13.700446Z","shell.execute_reply.started":"2024-06-13T05:58:06.571341Z","shell.execute_reply":"2024-06-13T05:59:13.699594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall numpy -y\n!pip install numpy==1.24.0\n!pip install facenet-pytorch\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:04:39.210473Z","iopub.execute_input":"2024-06-13T06:04:39.210846Z","iopub.status.idle":"2024-06-13T06:04:48.114536Z","shell.execute_reply.started":"2024-06-13T06:04:39.210792Z","shell.execute_reply":"2024-06-13T06:04:48.113653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install facenet-pytorch\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:48:27.902442Z","iopub.execute_input":"2024-06-13T06:48:27.902753Z","iopub.status.idle":"2024-06-13T06:48:32.999564Z","shell.execute_reply.started":"2024-06-13T06:48:27.902697Z","shell.execute_reply":"2024-06-13T06:48:32.998554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall torch torchvision numpy -y\n!pip install torch==1.7.1 torchvision==0.8.2 numpy==1.19.5\n!pip install facenet-pytorch==2.5.2\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:53:33.964058Z","iopub.execute_input":"2024-06-13T06:53:33.964454Z","iopub.status.idle":"2024-06-13T06:54:58.570117Z","shell.execute_reply.started":"2024-06-13T06:53:33.964406Z","shell.execute_reply":"2024-06-13T06:54:58.569282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npip install torch torchvision\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:56:44.20977Z","iopub.execute_input":"2024-06-13T05:56:44.210118Z","iopub.status.idle":"2024-06-13T05:56:50.126097Z","shell.execute_reply.started":"2024-06-13T05:56:44.210059Z","shell.execute_reply":"2024-06-13T05:56:50.125096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\nimport numpy as np\nimport pandas as pd\nimport matplotlib\nimport seaborn as sns\nimport torch\nimport matplotlib.pyplot as plt\n# from tqdm import tqdm_notebook\n%matplotlib inline \n# from google.colab.patches import cv2_imshow\nfrom IPython.display import HTML #imports to play videos\nfrom base64 import b64encode \nimport cv2 as cv\n#from skimage.measure import compare_ssim\nfrom skimage.metrics import structural_similarity as compare_ssim\n#from albumentations.pytorch.transforms import ToTensor\nfrom albumentations.pytorch import ToTensorV2 as ToTensor\n\nimport glob\nimport time\nfrom PIL import Image\nfrom facenet_pytorch import MTCNN, InceptionResnetV1, extract_face\nfrom tqdm import tqdm\n\nimport math\nimport pickle\nfrom functools import partial\nfrom collections import defaultdict\n\nfrom PIL import Image\nfrom glob import glob\n\nimport cv2\nimport skimage.measure\n#import albumentations as A\n#from albumentations.pytorch import ToTensor\n\nfrom tqdm.notebook import tqdm \nfrom albumentations.pytorch import ToTensor \n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\nfrom torchvision.models.video import mc3_18, r2plus1d_18\n\nfrom facenet_pytorch import MTCNN","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:06.623434Z","iopub.execute_input":"2024-06-13T06:55:06.624116Z","iopub.status.idle":"2024-06-13T06:55:06.825242Z","shell.execute_reply.started":"2024-06-13T06:55:06.623825Z","shell.execute_reply":"2024-06-13T06:55:06.824621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade albumentations\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:06:51.954693Z","iopub.execute_input":"2024-06-13T06:06:51.954991Z","iopub.status.idle":"2024-06-13T06:06:58.847331Z","shell.execute_reply.started":"2024-06-13T06:06:51.954948Z","shell.execute_reply":"2024-06-13T06:06:58.846373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = \"../input/deepfake-detection-challenge\" \nTRAIN_SAMPLE_FOLDER = \"train_sample_videos\"\nTEST_FOLDER = \"test_videos\"","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:18.965236Z","iopub.execute_input":"2024-06-13T06:55:18.965598Z","iopub.status.idle":"2024-06-13T06:55:18.969668Z","shell.execute_reply.started":"2024-06-13T06:55:18.965555Z","shell.execute_reply":"2024-06-13T06:55:18.968823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '../input/haarcascades'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")    ","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:21.313886Z","iopub.execute_input":"2024-06-13T06:55:21.314343Z","iopub.status.idle":"2024-06-13T06:55:21.32552Z","shell.execute_reply.started":"2024-06-13T06:55:21.314142Z","shell.execute_reply":"2024-06-13T06:55:21.324498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\") ","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:25.554105Z","iopub.execute_input":"2024-06-13T06:55:25.554421Z","iopub.status.idle":"2024-06-13T06:55:25.894417Z","shell.execute_reply.started":"2024-06-13T06:55:25.554375Z","shell.execute_reply":"2024-06-13T06:55:25.893643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:30.064343Z","iopub.execute_input":"2024-06-13T06:55:30.064684Z","iopub.status.idle":"2024-06-13T06:55:30.253998Z","shell.execute_reply.started":"2024-06-13T06:55:30.064637Z","shell.execute_reply":"2024-06-13T06:55:30.25317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")\n#reading the json file\ndef get_meta_from_json(path):\n    df = pd.read_json(os.path.join(DATA_FOLDER, path, json_file))\n    df = df.T\n    return df\n\nmeta_train_df = get_meta_from_json(TRAIN_SAMPLE_FOLDER)\nmeta_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:32.286308Z","iopub.execute_input":"2024-06-13T06:55:32.286669Z","iopub.status.idle":"2024-06-13T06:55:32.777956Z","shell.execute_reply.started":"2024-06-13T06:55:32.286624Z","shell.execute_reply":"2024-06-13T06:55:32.777048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:48.172175Z","iopub.execute_input":"2024-06-13T06:55:48.172559Z","iopub.status.idle":"2024-06-13T06:55:48.180236Z","shell.execute_reply.started":"2024-06-13T06:55:48.172505Z","shell.execute_reply":"2024-06-13T06:55:48.179507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:55:52.16095Z","iopub.execute_input":"2024-06-13T06:55:52.161294Z","iopub.status.idle":"2024-06-13T06:55:52.204928Z","shell.execute_reply.started":"2024-06-13T06:55:52.161232Z","shell.execute_reply":"2024-06-13T06:55:52.20393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df.loc[meta_train_df.label == 'REAL'])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:26.91448Z","iopub.execute_input":"2024-06-13T06:56:26.914773Z","iopub.status.idle":"2024-06-13T06:56:26.931025Z","shell.execute_reply.started":"2024-06-13T06:56:26.914737Z","shell.execute_reply":"2024-06-13T06:56:26.930115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Totals']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique() #collect all unique instances\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:31.34666Z","iopub.execute_input":"2024-06-13T06:56:31.347008Z","iopub.status.idle":"2024-06-13T06:56:31.353735Z","shell.execute_reply.started":"2024-06-13T06:56:31.346947Z","shell.execute_reply":"2024-06-13T06:56:31.352879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:35.238936Z","iopub.execute_input":"2024-06-13T06:56:35.239272Z","iopub.status.idle":"2024-06-13T06:56:35.253875Z","shell.execute_reply.started":"2024-06-13T06:56:35.239216Z","shell.execute_reply":"2024-06-13T06:56:35.253174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_frequent_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    items = []\n    vals = []\n    for col in data.columns:\n        itm = data[col].value_counts().index[0]\n        val = data[col].value_counts().values[0]\n        items.append(itm)\n        vals.append(val)\n    tt['Most frequent item'] = items\n    tt['Frequence'] = vals\n    tt['Percent from total'] = np.round(vals / total * 100, 3)\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:39.413271Z","iopub.execute_input":"2024-06-13T06:56:39.413647Z","iopub.status.idle":"2024-06-13T06:56:39.421718Z","shell.execute_reply.started":"2024-06-13T06:56:39.413584Z","shell.execute_reply":"2024-06-13T06:56:39.420822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:43.486448Z","iopub.execute_input":"2024-06-13T06:56:43.486752Z","iopub.status.idle":"2024-06-13T06:56:43.511143Z","shell.execute_reply.started":"2024-06-13T06:56:43.486711Z","shell.execute_reply":"2024-06-13T06:56:43.510231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df.loc[meta_train_df.label == 'FAKE'])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:48.456137Z","iopub.execute_input":"2024-06-13T06:56:48.456455Z","iopub.status.idle":"2024-06-13T06:56:48.482365Z","shell.execute_reply.started":"2024-06-13T06:56:48.456409Z","shell.execute_reply":"2024-06-13T06:56:48.481622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef plot_count(feature, title, df, size=1):\n    '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n    '''  \n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    \n    # Convert the feature to a categorical data type if it's not already\n    if not isinstance(df[feature].dtype, pd.CategoricalDtype):\n        df[feature] = df[feature].astype('category')\n    \n    # Ensure the feature is a Pandas Series\n    feature_series = df[feature]\n    \n    # Generate count plot\n    g = sns.countplot(x=feature_series, order=feature_series.value_counts().index[:20], palette='Set3')\n    g.set_title(f\"Number and percentage of {title}\")\n\n    if size > 2:\n        plt.xticks(rotation=90, size=8)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x() + p.get_width() / 2., height + 3, '{:1.2f}%'.format(100 * height / total), ha=\"center\")\n\n    plt.show()\n\n# Example usage to plot the distribution of the 'split' feature\nplot_count('split', 'split(train)', meta_train_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:56:58.850707Z","iopub.execute_input":"2024-06-13T06:56:58.851039Z","iopub.status.idle":"2024-06-13T06:56:59.099501Z","shell.execute_reply.started":"2024-06-13T06:56:58.850992Z","shell.execute_reply":"2024-06-13T06:56:59.098399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = np.array(list(meta_train_df.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:03.602479Z","iopub.execute_input":"2024-06-13T06:57:03.60279Z","iopub.status.idle":"2024-06-13T06:57:03.611073Z","shell.execute_reply.started":"2024-06-13T06:57:03.602747Z","shell.execute_reply":"2024-06-13T06:57:03.610231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='FAKE'].sample(3).index)\nfake_train_sample_video\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:07.798987Z","iopub.execute_input":"2024-06-13T06:57:07.799304Z","iopub.status.idle":"2024-06-13T06:57:07.808203Z","shell.execute_reply.started":"2024-06-13T06:57:07.799258Z","shell.execute_reply":"2024-06-13T06:57:07.80738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_img = cv.VideoCapture(video_path)\n    ret, frame = capture_img.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:10.194145Z","iopub.execute_input":"2024-06-13T06:57:10.194541Z","iopub.status.idle":"2024-06-13T06:57:10.200862Z","shell.execute_reply.started":"2024-06-13T06:57:10.194477Z","shell.execute_reply":"2024-06-13T06:57:10.200028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video_file))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:14.646786Z","iopub.execute_input":"2024-06-13T06:57:14.647141Z","iopub.status.idle":"2024-06-13T06:57:16.618398Z","shell.execute_reply.started":"2024-06-13T06:57:14.647076Z","shell.execute_reply":"2024-06-13T06:57:16.617448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='REAL'].sample(3).index) #viewing the real videos\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:20.170544Z","iopub.execute_input":"2024-06-13T06:57:20.170879Z","iopub.status.idle":"2024-06-13T06:57:20.179537Z","shell.execute_reply.started":"2024-06-13T06:57:20.170822Z","shell.execute_reply":"2024-06-13T06:57:20.178818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video in real_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:26.06384Z","iopub.execute_input":"2024-06-13T06:57:26.064195Z","iopub.status.idle":"2024-06-13T06:57:28.059823Z","shell.execute_reply.started":"2024-06-13T06:57:26.064133Z","shell.execute_reply":"2024-06-13T06:57:28.059055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    #we only show images extracted from first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n      video_path = os.path.join(DATA_FOLDER, video_folder, video_file)\n      capture_img = cv.VideoCapture(video_path)\n      ret, frame = capture_img.read()\n      frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n      ax[i//3, i%3].imshow(frame)\n      ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n      ax[i//3, i%3].axis('on')","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:32.420832Z","iopub.execute_input":"2024-06-13T06:57:32.42118Z","iopub.status.idle":"2024-06-13T06:57:32.430236Z","shell.execute_reply.started":"2024-06-13T06:57:32.42112Z","shell.execute_reply":"2024-06-13T06:57:32.429106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='meawmsgiti.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:36.947405Z","iopub.execute_input":"2024-06-13T06:57:36.947775Z","iopub.status.idle":"2024-06-13T06:57:38.753209Z","shell.execute_reply.started":"2024-06-13T06:57:36.947713Z","shell.execute_reply":"2024-06-13T06:57:38.75227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvf /kaggle/input/ffmpeg-static-build/ffmpeg-git-amd64-static.tar.xz","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:41.261391Z","iopub.execute_input":"2024-06-13T06:57:41.26171Z","iopub.status.idle":"2024-06-13T06:57:45.694688Z","shell.execute_reply.started":"2024-06-13T06:57:41.26167Z","shell.execute_reply":"2024-06-13T06:57:45.693832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob, shutil\nimport timeit, os, gc\nimport subprocess as sp\nfrom tqdm import tqdm\nfrom collections import defaultdict\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nimport json\nfrom IPython.display import HTML\nfrom base64 import b64encode\nimport cv2\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:50.267611Z","iopub.execute_input":"2024-06-13T06:57:50.267945Z","iopub.status.idle":"2024-06-13T06:57:50.279225Z","shell.execute_reply.started":"2024-06-13T06:57:50.267897Z","shell.execute_reply":"2024-06-13T06:57:50.278402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOME = \"./\"\nFFMPEG = \"/kaggle/working/ffmpeg-git-20191209-amd64-static\"\nFFMPEG_PATH = FFMPEG\nDATA_FOLDER = \"/kaggle/input/deepfake-detection-challenge\"\nTMP_FOLDER = HOME\nDATA_FOLDER_TRAIN = DATA_FOLDER\nVIDEOS_FOLDER_TRAIN = DATA_FOLDER_TRAIN + \"/train_sample_videos\"\nIMAGES_FOLDER_TRAIN = TMP_FOLDER + \"/images\"\nAUDIOS_FOLDER_TRAIN = TMP_FOLDER + \"/audios\"\nEXTRACT_META = True # False\nEXTRACT_CONTENT = True # False\nEXTRACT_FACES = True # False\nFRAME_RATE = 0.5 # Frame per\n\n# Ensure directories exist\nos.makedirs(IMAGES_FOLDER_TRAIN, exist_ok=True)\nos.makedirs(AUDIOS_FOLDER_TRAIN, exist_ok=True)\nprint(FFMPEG)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:54.36972Z","iopub.execute_input":"2024-06-13T06:57:54.370017Z","iopub.status.idle":"2024-06-13T06:57:54.377735Z","shell.execute_reply.started":"2024-06-13T06:57:54.369976Z","shell.execute_reply":"2024-06-13T06:57:54.376893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_command(*popenargs, **kwargs):\n    process = sp.Popen(stdout=sp.PIPE, stderr=sp.PIPE, *popenargs, **kwargs)\n    output, error = process.communicate()\n    retcode = process.poll()\n    \n    if retcode:\n        cmd = kwargs.get(\"args\")\n        if cmd is None:\n            cmd = popenargs[0]\n        error = sp.CalledProcessError(retcode, cmd)\n        error.output = output\n        raise error\n    return output\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:57:57.734347Z","iopub.execute_input":"2024-06-13T06:57:57.734713Z","iopub.status.idle":"2024-06-13T06:57:57.741842Z","shell.execute_reply.started":"2024-06-13T06:57:57.734654Z","shell.execute_reply":"2024-06-13T06:57:57.740922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ffprobe(filename, options=[\"-show_error\", \"-show_format\", \"-show_streams\", \"-show_programs\", \"-show_chapters\", \"-show_private_data\"]):\n    command = [FFMPEG_PATH + \"/ffprobe\", \"-v\", \"error\", *options, \"-print_format\", \"json\", filename]\n    ret = run_command(command)\n    if ret:\n        ret = json.loads(ret)\n    return ret\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:58:00.901598Z","iopub.execute_input":"2024-06-13T06:58:00.901915Z","iopub.status.idle":"2024-06-13T06:58:00.907923Z","shell.execute_reply.started":"2024-06-13T06:58:00.901873Z","shell.execute_reply":"2024-06-13T06:58:00.907137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def run_command(*popenargs, **kwargs):\n#     closeNULL = 0\n#     try:\n#         from subprocess import DEVNULL\n#         closeNULL = 0\n#     except ImportError:\n#         import os\n#         DEVNULL = open(os.devnull, 'wb')\n#         closeNULL = 1\n\n#     process = sp.Popen(stdout=sp.PIPE, stderr=DEVNULL, *popenargs, **kwargs)\n#     output, unused_err = process.communicate()\n#     retcode = process.poll()\n\n#     if closeNULL:\n#         DEVNULL.close()\n\n#     if retcode:\n#         cmd = kwargs.get(\"args\")\n#         if cmd is None:\n#             cmd = popenargs[0]\n#         error = sp.CalledProcessError(retcode, cmd)\n#         error.output = output\n#         raise error\n#     return output\n\n# def ffprobe(filename, options = [\"-show_error\", \"-show_format\", \"-show_streams\", \"-show_programs\", \"-show_chapters\", \"-show_private_data\"]):\n#     ret = {}\n#     command = [FFMPEG_PATH + \"/ffprobe\", \"-v\", \"error\", *options, \"-print_format\", \"json\", filename]\n#     ret = run_command(command)\n#     if ret:\n#         ret = json.loads(ret)\n#     return ret\n\n# ffmpeg -i input.mov -r 0.25 output_%04d.png\ndef ffextract_frames(filename, output_folder, rate = 0.25):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-r\", str(rate), \"-y\", output_folder + \"/output_%04d.png\"]\n    ret = run_command(command)\n    return ret\n\n# ffmpeg -i input-video.mp4 output-audio.mp3\ndef ffextract_audio(filename, output_path):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-vn\", \"-ac\", \"1\", \"-acodec\", \"copy\", \"-y\", output_path]\n    ret = run_command(command)\n    return ret","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:58:04.223048Z","iopub.execute_input":"2024-06-13T06:58:04.223401Z","iopub.status.idle":"2024-06-13T06:58:04.231092Z","shell.execute_reply.started":"2024-06-13T06:58:04.223329Z","shell.execute_reply":"2024-06-13T06:58:04.230174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTRACT_META:\n    results = []\n    filepaths = glob.glob(VIDEOS_FOLDER_TRAIN + \"/*.mp4\")\n    \n    for filepath in tqdm(filepaths):\n        js = ffprobe(filepath)\n        if js:\n            results.append(\n                (\n                    js.get(\"format\", {}).get(\"filename\").split(\"/\")[-1],\n                    js.get(\"format\", {}).get(\"format_long_name\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"codec_name\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"height\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"width\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"nb_frames\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"bit_rate\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"duration\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"start_time\"),\n                    js.get(\"streams\", [{}, {}])[0].get(\"avg_frame_rate\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"codec_name\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"channels\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"sample_rate\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"nb_frames\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"bit_rate\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"duration\"),\n                    js.get(\"streams\", [{}, {}])[1].get(\"start_time\")\n                )\n            )\n    \n    meta_df = pd.DataFrame(results, columns=[\n        \"filename\", \"format\", \"video_codec_name\", \"video_height\", \"video_width\",\n        \"video_nb_frames\", \"video_bit_rate\", \"video_duration\", \"video_start_time\", \"video_fps\",\n        \"audio_codec_name\", \"audio_channels\", \"audio_sample_rate\", \"audio_nb_frames\",\n        \"audio_bit_rate\", \"audio_duration\", \"audio_start_time\"\n    ])\n    \n    # Convert types for easier manipulation\n    meta_df[\"video_fps\"] = meta_df[\"video_fps\"].apply(lambda x: float(x.split(\"/\")[0]) / float(x.split(\"/\")[1]) if len(x.split(\"/\")) == 2 else None)\n    meta_df[\"video_duration\"] = meta_df[\"video_duration\"].astype(np.float32)\n    meta_df[\"video_bit_rate\"] = meta_df[\"video_bit_rate\"].astype(np.float32)\n    meta_df[\"video_start_time\"] = meta_df[\"video_start_time\"].astype(np.float32)\n    meta_df[\"video_nb_frames\"] = meta_df[\"video_nb_frames\"].astype(np.float32)\n    meta_df[\"audio_sample_rate\"] = meta_df[\"audio_sample_rate\"].astype(np.float32)\n    meta_df[\"audio_nb_frames\"] = meta_df[\"audio_nb_frames\"].astype(np.float32)\n    meta_df[\"audio_bit_rate\"] = meta_df[\"audio_bit_rate\"].astype(np.float32)\n    meta_df[\"audio_duration\"] = meta_df[\"audio_duration\"].astype(np.float32)\n    meta_df[\"audio_start_time\"] = meta_df[\"audio_start_time\"].astype(np.float32)\n    \n    meta_df.to_pickle(HOME + \"videos_meta.pkl\")\nelse:\n    meta_df = pd.read_pickle(HOME + \"videos_meta.pkl\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:58:08.465539Z","iopub.execute_input":"2024-06-13T06:58:08.465917Z","iopub.status.idle":"2024-06-13T06:58:29.840564Z","shell.execute_reply.started":"2024-06-13T06:58:08.465855Z","shell.execute_reply":"2024-06-13T06:58:29.839586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTRACT_META == True:\n    results = []\n    subfolder = VIDEOS_FOLDER_TRAIN\n    filepaths = glob.glob(subfolder + \"/*.mp4\")\n    for filepath in tqdm(filepaths):\n        js = ffprobe(filepath)\n#        print(js)\n        if js:\n            results.append(\n                (js.get(\"format\", {}).get(\"filename\")[len(subfolder) + 1:],\n                js.get(\"format\", {}).get(\"format_long_name\"),\n                # Video \n                js.get(\"streams\", [{}, {}])[0].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"height\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"width\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"start_time\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"avg_frame_rate\"),\n                 # Audio\n                js.get(\"streams\", [{}, {}])[1].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"channels\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"sample_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"start_time\")),\n            )\n\n    meta_pd = pd.DataFrame(results, columns=[\"filename\", \"format\", \"video_codec_name\", \"video_height\", \"video_width\",\n                                            \"video_nb_frames\", \"video_bit_rate\", \"video_duration\", \"video_start_time\",\"video_fps\",\n                                            \"audio_codec_name\", \"audio_channels\", \"audio_sample_rate\", \"audio_nb_frames\",\n                                            \"audio_bit_rate\", \"audio_duration\", \"audio_start_time\"])\n    meta_pd[\"video_fps\"] = meta_pd[\"video_fps\"].apply(lambda x: float(x.split(\"/\")[0])/float(x.split(\"/\")[1]) if len(x.split(\"/\")) == 2 else None)\n    meta_pd[\"video_duration\"] = meta_pd[\"video_duration\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"video_start_time\"] = meta_pd[\"video_start_time\"].astype(np.float32)\n    meta_pd[\"video_nb_frames\"] = meta_pd[\"video_nb_frames\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_sample_rate\"] = meta_pd[\"audio_sample_rate\"].astype(np.float32)\n    meta_pd[\"audio_nb_frames\"] = meta_pd[\"audio_nb_frames\"].astype(np.float32)\n    meta_pd[\"audio_bit_rate\"] = meta_pd[\"audio_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_duration\"] = meta_pd[\"audio_duration\"].astype(np.float32)\n    meta_pd[\"audio_start_time\"] = meta_pd[\"audio_start_time\"].astype(np.float32)\n    meta_pd.to_pickle(HOME + \"videos_meta.pkl\")\nelse:\n    meta_pd = pd.read_pickle(HOME + \"videos_meta.pkl\")\nmeta_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:59:59.196801Z","iopub.execute_input":"2024-06-13T06:59:59.197132Z","iopub.status.idle":"2024-06-13T07:00:13.330603Z","shell.execute_reply.started":"2024-06-13T06:59:59.19709Z","shell.execute_reply":"2024-06-13T07:00:13.329701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,6, figsize=(22, 3))\nd = sns.distplot(meta_pd[\"video_fps\"], ax=ax[0])\nd = sns.distplot(meta_pd[\"video_duration\"], ax=ax[1])\nd = sns.distplot(meta_pd[\"video_width\"], ax=ax[2])\nd = sns.distplot(meta_pd[\"video_height\"], ax=ax[3])\nd = sns.distplot(meta_pd[\"video_nb_frames\"], ax=ax[4])\nd = sns.distplot(meta_pd[\"video_bit_rate\"], ax=ax[5])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:00:17.161378Z","iopub.execute_input":"2024-06-13T07:00:17.161735Z","iopub.status.idle":"2024-06-13T07:00:18.885079Z","shell.execute_reply.started":"2024-06-13T07:00:17.161672Z","shell.execute_reply":"2024-06-13T07:00:18.883992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T06:59:29.042964Z","iopub.execute_input":"2024-06-13T06:59:29.043326Z","iopub.status.idle":"2024-06-13T06:59:29.33227Z","shell.execute_reply.started":"2024-06-13T06:59:29.043266Z","shell.execute_reply":"2024-06-13T06:59:29.331468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:00:27.158002Z","iopub.execute_input":"2024-06-13T07:00:27.158313Z","iopub.status.idle":"2024-06-13T07:00:27.334275Z","shell.execute_reply.started":"2024-06-13T07:00:27.158268Z","shell.execute_reply":"2024-06-13T07:00:27.333446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.merge(train_pd, meta_pd[[\"filename\", \"video_height\", \"video_width\", \"video_nb_frames\", \"video_bit_rate\", \"audio_nb_frames\"]], on=\"filename\", how=\"left\")\ntrain_pd[\"count\"] = train_pd.groupby([\"original\"])[\"original\"].transform('count')\n# train_pd.to_pickle(HOME + \"train_meta.pkl\")\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:00:30.942745Z","iopub.execute_input":"2024-06-13T07:00:30.943048Z","iopub.status.idle":"2024-06-13T07:00:30.977174Z","shell.execute_reply.started":"2024-06-13T07:00:30.943006Z","shell.execute_reply":"2024-06-13T07:00:30.976375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd.tail()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:00:34.697464Z","iopub.execute_input":"2024-06-13T07:00:34.697782Z","iopub.status.idle":"2024-06-13T07:00:34.714379Z","shell.execute_reply.started":"2024-06-13T07:00:34.69774Z","shell.execute_reply":"2024-06-13T07:00:34.713548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_df = pd.merge(train_df, meta_df[[\"filename\", \"video_height\", \"video_width\", \"video_nb_frames\", \"video_bit_rate\", \"audio_nb_frames\"]], on=\"filename\", how=\"left\")\ntrain_df[\"count\"] = train_df.groupby([\"original\"])[\"original\"].transform('count')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:00:50.214739Z","iopub.execute_input":"2024-06-13T07:00:50.21504Z","iopub.status.idle":"2024-06-13T07:00:50.392822Z","shell.execute_reply.started":"2024-06-13T07:00:50.214997Z","shell.execute_reply":"2024-06-13T07:00:50.392177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mtcnn","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:00:54.278753Z","iopub.execute_input":"2024-06-13T07:00:54.27908Z","iopub.status.idle":"2024-06-13T07:01:01.033548Z","shell.execute_reply.started":"2024-06-13T07:00:54.279036Z","shell.execute_reply":"2024-06-13T07:01:01.032458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUDIO_FORMAT = \"aac\" # \"wav\"\nvideos_folder = VIDEOS_FOLDER_TRAIN\nimages_folder_path = IMAGES_FOLDER_TRAIN\naudios_folder_path = AUDIOS_FOLDER_TRAIN\nif EXTRACT_CONTENT == True:\n    # 1h20min for chunk#0 (11GB)\n    # Extract some images + audio track\n    for idx, row in tqdm(train_pd.iterrows(), total=meta_pd.shape[0]):\n        try:\n            video_path = videos_folder + \"/\" + row[\"filename\"]\n            images_path = images_folder_path + \"/\" + row[\"filename\"][:-4]\n            audio_path = audios_folder_path + \"/\" + row[\"filename\"][:-4]\n            # Extract images\n            if not os.path.exists(images_path): os.makedirs(images_path)\n            ret = ffextract_frames(video_path, images_path, rate = FRAME_RATE)\n            # Extract audio\n            if not os.path.exists(audio_path): os.makedirs(audio_path)\n            # ret = ffextract_audio(video_path, audio_path + \"/audio.\" + AUDIO_FORMAT)\n        except:\n            print(\"Cannot extract frames/audio for:\" + row[\"filename\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:01:04.484899Z","iopub.execute_input":"2024-06-13T07:01:04.485287Z","iopub.status.idle":"2024-06-13T07:13:02.628633Z","shell.execute_reply.started":"2024-06-13T07:01:04.485207Z","shell.execute_reply":"2024-06-13T07:13:02.62775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 12\nfake = train_pd[\"filename\"][idx]\nreal = train_pd[\"original\"][idx]\nvid_width = train_pd[\"video_width\"][idx]\nvid_real = open(VIDEOS_FOLDER_TRAIN + \"/\" + real, 'rb').read()\ndata_url_real = \"data:video/mp4;base64,\" + b64encode(vid_real).decode()\nvid_fake = open(VIDEOS_FOLDER_TRAIN + \"/\" + fake, 'rb').read()\ndata_url_fake = \"data:video/mp4;base64,\" + b64encode(vid_fake).decode()\nHTML(\"\"\"\n<div style='width: 100%%; display: table;'>\n    <div style='display: table-row'>\n        <div style='width: %dpx; display: table-cell;'><b>Real</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n        <div style='display: table-cell;'><b>Fake</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n    </div>\n</div>\n\"\"\" % ( int(vid_width/3.2) + 10, \n       real, int(vid_width/3.2), data_url_real, \n       fake, int(vid_width/3.2), data_url_fake))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:14:40.032573Z","iopub.execute_input":"2024-06-13T07:14:40.032968Z","iopub.status.idle":"2024-06-13T07:14:40.162914Z","shell.execute_reply.started":"2024-06-13T07:14:40.032912Z","shell.execute_reply":"2024-06-13T07:14:40.161641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + \"haarcascade_frontalface_default.xml\")\n\ndef detect_face_cv2(img):\n    # Move to grayscale\n    gray_img = cv2.cvtColor(img.copy(), cv2.COLOR_RGB2GRAY)\n    face_locations = []\n    face_rects = face_cascade.detectMultiScale(gray_img, scaleFactor=1.3, minNeighbors=5)     \n    for (x,y,w,h) in face_rects: \n        face_location = (x,y,w,h)\n        face_locations.append((face_location, 1.0))\n    return face_locations","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:14:48.488641Z","iopub.execute_input":"2024-06-13T07:14:48.489083Z","iopub.status.idle":"2024-06-13T07:14:48.52614Z","shell.execute_reply.started":"2024-06-13T07:14:48.488899Z","shell.execute_reply":"2024-06-13T07:14:48.525479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mtcnn","metadata":{"execution":{"iopub.status.busy":"2024-06-13T05:44:41.890186Z","iopub.status.idle":"2024-06-13T05:44:41.890669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mtcnn import MTCNN\ndetector = MTCNN()\n\ndef detect_face_mtcnn(img):\n    face_locations = []\n    items = detector.detect_faces(img)\n    for face in items:\n        face_location = tuple(face.get('box'))\n        face_confidence = float(face.get('confidence'))\n        face_locations.append((face_location, face_confidence))\n    return face_locations","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:14:54.241422Z","iopub.execute_input":"2024-06-13T07:14:54.241749Z","iopub.status.idle":"2024-06-13T07:15:02.972632Z","shell.execute_reply.started":"2024-06-13T07:14:54.241708Z","shell.execute_reply":"2024-06-13T07:15:02.971879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_faces(files, source, detector=detect_face_cv2):\n    results = []\n    # for idx, file in tqdm(enumerate(files), total=len(files)):\n    for idx, file in enumerate(files):\n        try:\n            img = cv2.cvtColor(cv2.imread(file, cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)\n            face_locations = detector(img)\n            results.append((source, file[file.find(\"output_\"):], face_locations, len(face_locations)))\n        except:\n            print(\"Cannot extract faces for image: %s\" % file)\n    return results","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:15:07.920601Z","iopub.execute_input":"2024-06-13T07:15:07.920908Z","iopub.status.idle":"2024-06-13T07:15:07.927958Z","shell.execute_reply.started":"2024-06-13T07:15:07.920865Z","shell.execute_reply":"2024-06-13T07:15:07.927012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = fake\ndump_folder = IMAGES_FOLDER_TRAIN + \"/\" + file[:-4]\nfiles = glob.glob(dump_folder + \"/*\")\nDETECTORS = {\n    \"cv2\": detect_face_cv2,\n    \"mtcnn\": detect_face_mtcnn\n}\nfaces_pd = None\nfor key, value in DETECTORS.items():\n    tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n    if faces_pd is None:\n        faces_pd = tmp_pd\n    else:\n        faces_pd = pd.merge(faces_pd, tmp_pd, on=[\"filename\", \"image\"], how=\"left\")\nfaces_pd.head(12)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:15:12.271614Z","iopub.execute_input":"2024-06-13T07:15:12.271918Z","iopub.status.idle":"2024-06-13T07:15:20.036563Z","shell.execute_reply.started":"2024-06-13T07:15:12.271875Z","shell.execute_reply":"2024-06-13T07:15:20.035639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_faces_boxes(df, max_cols = 2, max_rows = 6, fsize=(24, 5), max_items=12):    \n    idx = 0    \n    for item_idx, item in df.iterrows():\n        img = cv2.cvtColor(cv2.imread(IMAGES_FOLDER_TRAIN + \"/\" + item[\"filename\"][:-4] +\"/\" + item[\"image\"], cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)    \n        face_img = img #.copy()\n        # grid subplots\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0: fig = plt.figure(figsize=fsize)\n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        # display image with boxes\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols, 0):\n            face_locations = item[c]\n            face_confidence = item[c]            \n            if len(face_locations) > 0:\n                for face_location in face_locations:        \n                    ((x,y,w,h), confidence) = face_location\n                    # face_img = face_img[y:y+h, x:x+w]\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255,i*255,0), 8)\n                    cv2.putText(face_img, '%.1f' % (confidence*100.0), (x+w, y+h), cv2.FONT_HERSHEY_SIMPLEX, 2.0, (255,i*255,0), 9, cv2.LINE_AA)\n                ax.imshow(face_img)\n            else:\n                ax.imshow(img)\n            ax.set_title(\"%s %s / %s - Faces: %d %s %s\" % (item[\"label\"] if \"label\" in df.columns else \"\", \n                                                           item[\"filename\"], item[\"image\"],\n                                                           item[\"faces_mtcnn\"] if \"faces_mtcnn\" in df.columns else len(face_locations),\n                                                           item[\"faces_mtcnn_median\"] if \"faces_mtcnn_median\" in df.columns else \"\",\n                                                           item[\"faces\"] if \"faces\" in df.columns else \"\"))\n        if (col == max_cols -1): plt.show()\n        idx = idx + 1\n        if (max_items > 0 and idx >=max_items): break","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:15:25.065868Z","iopub.execute_input":"2024-06-13T07:15:25.066204Z","iopub.status.idle":"2024-06-13T07:15:25.083777Z","shell.execute_reply.started":"2024-06-13T07:15:25.066158Z","shell.execute_reply":"2024-06-13T07:15:25.082938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(faces_pd)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:15:33.840936Z","iopub.execute_input":"2024-06-13T07:15:33.841292Z","iopub.status.idle":"2024-06-13T07:15:36.539369Z","shell.execute_reply.started":"2024-06-13T07:15:33.841234Z","shell.execute_reply":"2024-06-13T07:15:36.53843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_detector_on_video(videos_filename, verbose=False):\n    if verbose == True: \n        print(\"Starting with batch of %d videos\" % len(videos_filename))\n    tmp_faces_pd = None\n    for file in videos_filename:\n        # Find out dump folder with images\n        dump_folder = IMAGES_FOLDER_TRAIN + \"/\" + file[:-4]\n        # List files\n        files = glob.glob(dump_folder + \"/*\")\n        DETECTORS = {\n            \"mtcnn\": detect_face_mtcnn\n        }\n        for key, value in DETECTORS.items():\n            tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n            if tmp_faces_pd is None:\n                tmp_faces_pd = tmp_pd\n            else:\n                tmp_faces_pd = pd.concat([tmp_faces_pd, tmp_pd], axis=0)\n    return tmp_faces_pd","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:15:41.450313Z","iopub.execute_input":"2024-06-13T07:15:41.450704Z","iopub.status.idle":"2024-06-13T07:15:41.460332Z","shell.execute_reply.started":"2024-06-13T07:15:41.450643Z","shell.execute_reply":"2024-06-13T07:15:41.459492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\ncpus = multiprocessing.cpu_count()\nif EXTRACT_FACES == True:\n    resultfutures = []\n    results = []\n    tasks = np.array_split(train_pd[\"filename\"].unique(), 20)\n    print(\"Tasks: %d\" % len(tasks))\n    with ThreadPoolExecutor(max_workers=cpus) as executor:\n        resultfutures = tqdm(executor.map(run_detector_on_video, tasks), total=len(tasks))\n    results = [x for x in resultfutures]\n    executor.shutdown()\n    # Gather results\n    all_faces_pd = None\n    for result in results:\n        if all_faces_pd is None:\n            all_faces_pd = result\n        else:\n            all_faces_pd = pd.concat([all_faces_pd, result], axis=0)\n    all_faces_pd = all_faces_pd.reset_index(drop=True)\n    all_faces_pd.to_pickle(HOME + \"faces.pkl\")\nelse:\n    all_faces_pd = pd.read_pickle(HOME + \"faces.pkl\")\nprint(all_faces_pd.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:15:46.377446Z","iopub.execute_input":"2024-06-13T07:15:46.377813Z","iopub.status.idle":"2024-06-13T07:41:09.238305Z","shell.execute_reply.started":"2024-06-13T07:15:46.377752Z","shell.execute_reply":"2024-06-13T07:41:09.237588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_faces_pd[\"faces_mtcnn_avg\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmean)\nall_faces_pd[\"faces_mtcnn_median\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmedian)\nall_faces_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:46:12.300251Z","iopub.execute_input":"2024-06-13T07:46:12.300661Z","iopub.status.idle":"2024-06-13T07:46:12.332137Z","shell.execute_reply.started":"2024-06-13T07:46:12.3006Z","shell.execute_reply":"2024-06-13T07:46:12.331459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_avg\"], kde=True, ax=ax[0])\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_median\"], kde=False, ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:46:16.942296Z","iopub.execute_input":"2024-06-13T07:46:16.94265Z","iopub.status.idle":"2024-06-13T07:46:17.754317Z","shell.execute_reply.started":"2024-06-13T07:46:16.942589Z","shell.execute_reply":"2024-06-13T07:46:17.753176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade seaborn\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:47:02.332312Z","iopub.execute_input":"2024-06-13T07:47:02.33269Z","iopub.status.idle":"2024-06-13T07:47:08.997186Z","shell.execute_reply.started":"2024-06-13T07:47:02.332625Z","shell.execute_reply":"2024-06-13T07:47:08.996053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Create subplots\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\n\n# Plot distribution of faces_mtcnn_avg using distplot\nsns.distplot(all_faces_pd[\"faces_mtcnn_avg\"].dropna(), kde=True, ax=ax[0])\nax[0].set_title('Distribution of faces_mtcnn_avg')\n\n# Plot distribution of faces_mtcnn_median using distplot\nsns.distplot(all_faces_pd[\"faces_mtcnn_median\"].dropna(), kde=False, ax=ax[1])\nax[1].set_title('Distribution of faces_mtcnn_median')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:47:55.816683Z","iopub.execute_input":"2024-06-13T07:47:55.816977Z","iopub.status.idle":"2024-06-13T07:47:56.822074Z","shell.execute_reply.started":"2024-06-13T07:47:55.816935Z","shell.execute_reply":"2024-06-13T07:47:56.821217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Display summary statistics of the columns of interest\nprint(\"Summary statistics of faces_mtcnn_avg:\")\nprint(all_faces_pd[\"faces_mtcnn_avg\"].describe())\n\nprint(\"\\nSummary statistics of faces_mtcnn_median:\")\nprint(all_faces_pd[\"faces_mtcnn_median\"].describe())\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:08.866832Z","iopub.execute_input":"2024-06-13T07:48:08.867195Z","iopub.status.idle":"2024-06-13T07:48:08.881434Z","shell.execute_reply.started":"2024-06-13T07:48:08.867136Z","shell.execute_reply":"2024-06-13T07:48:08.880594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(all_faces_pd[all_faces_pd[\"faces_mtcnn\"] == 3], max_items=24)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:13.923821Z","iopub.execute_input":"2024-06-13T07:48:13.92412Z","iopub.status.idle":"2024-06-13T07:48:21.521719Z","shell.execute_reply.started":"2024-06-13T07:48:13.924076Z","shell.execute_reply":"2024-06-13T07:48:21.520809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_faces_pd = pd.merge(all_faces_pd, train_pd, on=\"filename\", how=\"left\")\nclean_faces_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:26.47761Z","iopub.execute_input":"2024-06-13T07:48:26.477926Z","iopub.status.idle":"2024-06-13T07:48:26.517173Z","shell.execute_reply.started":"2024-06-13T07:48:26.477882Z","shell.execute_reply":"2024-06-13T07:48:26.515617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def faces_max_item(boxes, idx1, idx2):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[idx1][idx2])[idx1][idx2]\n    return ret\n\ndef faces_max_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[1])[1]\n    return ret\n\ndef faces_min_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = min(boxes, key=lambda item: item[1])[1]\n    return ret\n\nclean_faces_pd[\"faces_max_width\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 2)) \nclean_faces_pd[\"faces_max_height\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 3))\nclean_faces_pd[\"faces_max_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_confidence(x))\nclean_faces_pd[\"faces_min_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_min_confidence(x))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:30.25215Z","iopub.execute_input":"2024-06-13T07:48:30.25247Z","iopub.status.idle":"2024-06-13T07:48:30.294261Z","shell.execute_reply.started":"2024-06-13T07:48:30.252423Z","shell.execute_reply":"2024-06-13T07:48:30.293384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Faces stats:\")\nprint(clean_faces_pd[[\"faces_max_width\", \"faces_max_height\", \"faces_min_conf\", \"faces_max_conf\"]].describe(percentiles=[0.01,0.05, 0.1,0.25,0.5,0.75,0.9,0.95,0.99]))\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_max_width\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_height\"], kde=True, ax=ax[1])\nplt.show()\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_min_conf\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_conf\"], kde=True, ax=ax[1])\nfig, ax = plt.subplots(figsize=(22, 3))\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_width\", y=\"faces_max_conf\", c=\"red\", ax=ax, label=\"faces_max_width\", alpha=0.5)\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_height\", y=\"faces_max_conf\", c=\"blue\", ax=d,  label=\"faces_max_height\", alpha=0.5)\nd = plt.legend(loc=\"upper right\")","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:34.542014Z","iopub.execute_input":"2024-06-13T07:48:34.54235Z","iopub.status.idle":"2024-06-13T07:48:36.536807Z","shell.execute_reply.started":"2024-06-13T07:48:34.542289Z","shell.execute_reply":"2024-06-13T07:48:36.535539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install imutils","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:41.304887Z","iopub.execute_input":"2024-06-13T07:48:41.305184Z","iopub.status.idle":"2024-06-13T07:48:49.711259Z","shell.execute_reply.started":"2024-06-13T07:48:41.305143Z","shell.execute_reply":"2024-06-13T07:48:49.710248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport shutil\nimport cv2\nimport pandas as pd\nimport matplotlib\nmatplotlib.use(\"Agg\")\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pickle\nfrom imutils import paths\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import VGG16\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dense\nfrom keras.layers import Input\nfrom keras.models import Model\nfrom keras.optimizers import SGD\nfrom sklearn.metrics import classification_report\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:53.976505Z","iopub.execute_input":"2024-06-13T07:48:53.976862Z","iopub.status.idle":"2024-06-13T07:48:54.285072Z","shell.execute_reply.started":"2024-06-13T07:48:53.976796Z","shell.execute_reply":"2024-06-13T07:48:54.284125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/working/finetuningkeras/dataset\"\n\n# define the names of the training, testing, and validation\n# directories\nTRAIN = \"training\"\nTEST = \"evaluation\"\nVAL = \"validation\"\n\nREAL = 'REAL'\nFAKE = 'FAKE'\n\n# initialize the list of class label names\nCLASSES = [\"FAKE\", \"REAL\"]\n\n\n# set the batch size when fine-tuning\nBATCH_SIZE = 32\n\ntrainEpochs = 10\nepochsFineTune = 10\nmaxVids = 5\n\n# set the path to the serialized model after training\nMODEL_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"Deepfake.model\"])\n\n# define the path to the output training history plots\nUNFROZEN_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"unfrozen.png\"])\nWARMUP_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"warmup.png\"])\n\nfile = '/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json'\nimg_path = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ndata_path = '/kaggle/working/finetuningkeras/real_fake'\ndir_fake_frames = '/kaggle/working/FAKE_frames'\ndir_real_frames = '/kaggle/working/REAL_frames'\ndir_output = '/kaggle/working/finetuningkeras/output'\n\ndir_data_path_real = os.path.join(data_path, REAL)\ndir_data_path_fake = os.path.join(data_path, FAKE)\n\ndir_train_real = os.path.join(BASE_PATH, TRAIN, REAL)\ndir_train_fake = os.path.join(BASE_PATH, TRAIN, FAKE)\ndir_valid_real = os.path.join(BASE_PATH, VAL, REAL)\ndir_valid_fake = os.path.join(BASE_PATH, VAL, FAKE)\ndir_test_real = os.path.join(BASE_PATH, TEST, REAL)\ndir_test_fake = os.path.join(BASE_PATH, TEST, FAKE)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:48:59.892476Z","iopub.execute_input":"2024-06-13T07:48:59.892816Z","iopub.status.idle":"2024-06-13T07:48:59.905685Z","shell.execute_reply.started":"2024-06-13T07:48:59.892756Z","shell.execute_reply":"2024-06-13T07:48:59.904735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = '/kaggle/working/finetuningkeras/real_fake/FAKE'\noutput_dir = '/kaggle/working/FAKE_frames/'\ndef explode_frames(input_dir, output_dir, maxN):\n\n    mp4_filenames = [f for f in os.listdir(input_dir) if f.endswith('.mp4')]\n    n = 0\n    \n    for mp4fn in mp4_filenames:\n        \n        if(n < maxN):\n            n += 1 \n            mp4fp = os.path.join(input_dir, mp4fn)\n            cam = cv2.VideoCapture(mp4fp) \n            if(cam.isOpened()):\n                print('Processing file #'+ str(n) + ' (' + mp4fn + ')...')\n            else: \n                print('Problem opening file #'+ str(n) + ' (' + mp4fn + ')...')\n                continue \n            \n            nframe = 0\n            while(True): #continue until ret = False then break\n                nframe += 1\n                ret,frame = cam.read()\n                \n                if ret: \n                    # if video is still left continue creating images \n                    out_filename = os.path.splitext(mp4fn)[0]+  '_frame' + str(nframe) + '.jpg'\n                    out_filepath =  os.path.join(output_dir, out_filename)\n                    \n                    # writing the extracted images \n                    cv2.imwrite(out_filepath, frame) \n                else: \n                    break\n\n            # Release all space and windows once done\n            print(' - created ' + str(nframe-1) + ' images') # -1 bc count incremented before exit\n            cam.release() \n            cv2.destroyAllWindows()\n            \n        else: \n            break\n            \n\"\"\"\nDistribute files/images from a source directory into training, validation, and testing directories. \nsrc_dir = source/input directory\ntrain_dir, val_dir, test_dir = target training/validation/testing directory\nvalperc = fraction of dataset to use for validation (0-1)\ntestperc = fraction of dataset to use for testing (0-1)\n\"\"\"\n            \ndef trainvaltest_split(src_dir, train_dir, val_dir, test_dir, valperc = 0.15, testperc = 0.15):\n    \n    filenames = os.listdir(src_dir) #get all filenames in random order\n    np.random.shuffle(filenames)\n    \n    n = len(filenames)\n    split1 = int(n*(1 - (valperc + testperc)))\n    split2 = int(n*(1 - (testperc)))\n    \n    fn_train, fn_val, fn_test = np.split(np.array(filenames), [split1, split2])\n    \n    fn_lists = [fn_train, fn_val, fn_test]\n    targetdirs = [train_dir, val_dir, test_dir]\n    \n    print('Total images: ', n)\n    print('Training: ', len(fn_train))\n    print('Validation: ', len(fn_val))\n    print('Testing: ', len(fn_test))\n    \n    all_fp = [os.path.join(src_dir, fn) for fn in filenames]\n    \n    #move files\n    for i, fn_list in enumerate(fn_lists):\n        for fn in fn_list: \n            target_dir = targetdirs[i]\n            fp_from = os.path.join(src_dir, fn)\n            fp_to = os.path.join(target_dir, fn)\n            \n            shutil.move(fp_from, fp_to)\n\n            \n\"\"\"\nConstruct a plot that plots and saves the training history\n\"\"\"           \ndef plot_training(H, N, plotPath):\n\tplt.style.use(\"ggplot\")\n\tplt.figure()\n\tplt.plot(np.arange(0, N), H.history[\"loss\"], label=\"train_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"val_loss\"], label=\"val_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"accuracy\"], label=\"train_acc\")\n\tplt.plot(np.arange(0, N), H.history[\"val_accuracy\"], label=\"val_acc\")\n\tplt.title(\"Training Loss and Accuracy\")\n\tplt.xlabel(\"Epoch #\")\n\tplt.ylabel(\"Loss/Accuracy\")\n\tplt.legend(loc=\"lower left\")\n\tplt.savefig(plotPath)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:49:06.568644Z","iopub.execute_input":"2024-06-13T07:49:06.568995Z","iopub.status.idle":"2024-06-13T07:49:06.597163Z","shell.execute_reply.started":"2024-06-13T07:49:06.568935Z","shell.execute_reply":"2024-06-13T07:49:06.596206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nos.makedirs(dir_train_real, exist_ok = True)\nos.makedirs(dir_train_fake, exist_ok = True)\nos.makedirs(dir_valid_real, exist_ok = True)\nos.makedirs(dir_valid_fake, exist_ok = True)\nos.makedirs(dir_test_real, exist_ok = True)\nos.makedirs(dir_test_fake, exist_ok = True)\n\nos.makedirs(dir_data_path_real, exist_ok = True)\nos.makedirs(dir_data_path_fake, exist_ok = True)\nos.makedirs(dir_fake_frames, exist_ok = True) \nos.makedirs(dir_real_frames, exist_ok = True) \nos.makedirs(dir_output, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:49:14.572332Z","iopub.execute_input":"2024-06-13T07:49:14.572788Z","iopub.status.idle":"2024-06-13T07:49:14.582213Z","shell.execute_reply.started":"2024-06-13T07:49:14.572604Z","shell.execute_reply":"2024-06-13T07:49:14.581468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_json(file)\ndf = df.T\n\n# %% [code]\nlabel = df[['label']]","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:49:19.817449Z","iopub.execute_input":"2024-06-13T07:49:19.817788Z","iopub.status.idle":"2024-06-13T07:49:19.990521Z","shell.execute_reply.started":"2024-06-13T07:49:19.817728Z","shell.execute_reply":"2024-06-13T07:49:19.989765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fn, row in label.iterrows():\n    src = os.path.join(img_path, fn)\n    dest = os.path.join(data_path, row['label'], fn)\n    shutil.copy(src, dest)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:49:24.717275Z","iopub.execute_input":"2024-06-13T07:49:24.717641Z","iopub.status.idle":"2024-06-13T07:49:28.114148Z","shell.execute_reply.started":"2024-06-13T07:49:24.717575Z","shell.execute_reply":"2024-06-13T07:49:28.113468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_fake, dir_fake_frames, maxN= maxVids)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:49:28.923026Z","iopub.execute_input":"2024-06-13T07:49:28.923365Z","iopub.status.idle":"2024-06-13T07:50:22.233289Z","shell.execute_reply.started":"2024-06-13T07:49:28.923303Z","shell.execute_reply":"2024-06-13T07:50:22.2324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_real, dir_real_frames, maxN= maxVids)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:50:22.235676Z","iopub.execute_input":"2024-06-13T07:50:22.236011Z","iopub.status.idle":"2024-06-13T07:51:15.250535Z","shell.execute_reply.started":"2024-06-13T07:50:22.235949Z","shell.execute_reply":"2024-06-13T07:51:15.249632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_fake_frames,\n                   train_dir = dir_train_fake, \n                   val_dir = dir_valid_fake, \n                   test_dir = dir_test_fake)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:51:15.251943Z","iopub.execute_input":"2024-06-13T07:51:15.252222Z","iopub.status.idle":"2024-06-13T07:51:15.313917Z","shell.execute_reply.started":"2024-06-13T07:51:15.252164Z","shell.execute_reply":"2024-06-13T07:51:15.313095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_real_frames,\n                   train_dir = dir_train_real, \n                   val_dir = dir_valid_real, \n                   test_dir = dir_test_real)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:51:15.315592Z","iopub.execute_input":"2024-06-13T07:51:15.31593Z","iopub.status.idle":"2024-06-13T07:51:15.376407Z","shell.execute_reply.started":"2024-06-13T07:51:15.315873Z","shell.execute_reply":"2024-06-13T07:51:15.37569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrainAug = ImageDataGenerator(\n\trotation_range=30,\n\tzoom_range=0.15,\n\twidth_shift_range=0.2,\n\theight_shift_range=0.2,\n\tshear_range=0.15,\n\thorizontal_flip=True,\n\tfill_mode=\"nearest\")","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:51:15.378532Z","iopub.execute_input":"2024-06-13T07:51:15.378761Z","iopub.status.idle":"2024-06-13T07:51:15.383226Z","shell.execute_reply.started":"2024-06-13T07:51:15.378723Z","shell.execute_reply":"2024-06-13T07:51:15.382475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valAug = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:51:15.384802Z","iopub.execute_input":"2024-06-13T07:51:15.385046Z","iopub.status.idle":"2024-06-13T07:51:15.392404Z","shell.execute_reply.started":"2024-06-13T07:51:15.385003Z","shell.execute_reply":"2024-06-13T07:51:15.391499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = np.array([123.68, 116.779, 103.939], dtype=\"float32\")\ntrainAug.mean = mean\nvalAug.mean = mean","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:02.250672Z","iopub.execute_input":"2024-06-13T07:52:02.250992Z","iopub.status.idle":"2024-06-13T07:52:02.255585Z","shell.execute_reply.started":"2024-06-13T07:52:02.250949Z","shell.execute_reply":"2024-06-13T07:52:02.254688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainPath = os.path.join(BASE_PATH, TRAIN)\ntrainGen = trainAug.flow_from_directory(\n\ttrainPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=True,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the validation generator\nvalPath = os.path.join(BASE_PATH, VAL)\nvalGen = valAug.flow_from_directory(\n\tvalPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the testing generator\ntestPath = os.path.join(BASE_PATH, TEST)\ntestGen = valAug.flow_from_directory(\n\ttestPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:05.619839Z","iopub.execute_input":"2024-06-13T07:52:05.620191Z","iopub.status.idle":"2024-06-13T07:52:05.942327Z","shell.execute_reply.started":"2024-06-13T07:52:05.620132Z","shell.execute_reply":"2024-06-13T07:52:05.941648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseModel = VGG16(weights=\"imagenet\", include_top=False,\n\tinput_tensor=Input(shape=(224, 224, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:10.347885Z","iopub.execute_input":"2024-06-13T07:52:10.348185Z","iopub.status.idle":"2024-06-13T07:52:12.20948Z","shell.execute_reply.started":"2024-06-13T07:52:10.34814Z","shell.execute_reply":"2024-06-13T07:52:12.208801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headModel = baseModel.output\nheadModel = Flatten(name=\"flatten\")(headModel)\nheadModel = Dense(512, activation=\"relu\")(headModel)\nheadModel = Dropout(0.5)(headModel)\nheadModel = Dense(len(CLASSES), activation=\"softmax\")(headModel)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:15.276908Z","iopub.execute_input":"2024-06-13T07:52:15.277207Z","iopub.status.idle":"2024-06-13T07:52:15.322287Z","shell.execute_reply.started":"2024-06-13T07:52:15.277167Z","shell.execute_reply":"2024-06-13T07:52:15.321489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=baseModel.input, outputs=headModel)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:22.789946Z","iopub.execute_input":"2024-06-13T07:52:22.790243Z","iopub.status.idle":"2024-06-13T07:52:22.795339Z","shell.execute_reply.started":"2024-06-13T07:52:22.790198Z","shell.execute_reply":"2024-06-13T07:52:22.794325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers:\n\tlayer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:28.12749Z","iopub.execute_input":"2024-06-13T07:52:28.127785Z","iopub.status.idle":"2024-06-13T07:52:28.131925Z","shell.execute_reply.started":"2024-06-13T07:52:28.127743Z","shell.execute_reply":"2024-06-13T07:52:28.130858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:38.885701Z","iopub.execute_input":"2024-06-13T07:52:38.885991Z","iopub.status.idle":"2024-06-13T07:52:38.93439Z","shell.execute_reply.started":"2024-06-13T07:52:38.885948Z","shell.execute_reply":"2024-06-13T07:52:38.933645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"totalTrain = len(list(paths.list_images(trainPath)))\ntotalVal = len(list(paths.list_images(valPath)))\ntotalTest = len(list(paths.list_images(testPath)))\n\nprint(\"[INFO] training head...\")\nH = model.fit(\n    trainGen,\n    steps_per_epoch=totalTrain // BATCH_SIZE,\n    validation_data=valGen,\n    validation_steps=totalVal // BATCH_SIZE,\n    epochs=trainEpochs)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:52:44.860309Z","iopub.execute_input":"2024-06-13T07:52:44.860664Z","iopub.status.idle":"2024-06-13T08:05:52.548932Z","shell.execute_reply.started":"2024-06-13T07:52:44.860615Z","shell.execute_reply":"2024-06-13T08:05:52.547836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network head...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\n\n\nplot_training(H, trainEpochs, WARMUP_PLOT_PATH)\nplt.show()  ","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:05:52.551058Z","iopub.execute_input":"2024-06-13T08:05:52.551388Z","iopub.status.idle":"2024-06-13T08:06:05.237878Z","shell.execute_reply.started":"2024-06-13T08:05:52.55132Z","shell.execute_reply":"2024-06-13T08:06:05.236518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:05.240104Z","iopub.execute_input":"2024-06-13T08:06:05.240603Z","iopub.status.idle":"2024-06-13T08:06:06.834028Z","shell.execute_reply.started":"2024-06-13T08:06:05.24052Z","shell.execute_reply":"2024-06-13T08:06:06.833181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns ","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:06.835761Z","iopub.execute_input":"2024-06-13T08:06:06.836005Z","iopub.status.idle":"2024-06-13T08:06:06.841089Z","shell.execute_reply.started":"2024-06-13T08:06:06.83596Z","shell.execute_reply":"2024-06-13T08:06:06.840138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming testGen and model are defined\ntestGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\ny_true = testGen.classes\n\n# Compute ROC curve for the positive class\nfpr, tpr, _ = roc_curve(y_true, y_pred_probs[:, 1])\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=1, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:06.845046Z","iopub.execute_input":"2024-06-13T08:06:06.845389Z","iopub.status.idle":"2024-06-13T08:06:18.942001Z","shell.execute_reply.started":"2024-06-13T08:06:06.84531Z","shell.execute_reply":"2024-06-13T08:06:18.940667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:18.944392Z","iopub.execute_input":"2024-06-13T08:06:18.944818Z","iopub.status.idle":"2024-06-13T08:06:30.783421Z","shell.execute_reply.started":"2024-06-13T08:06:18.944748Z","shell.execute_reply":"2024-06-13T08:06:30.782599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:30.785161Z","iopub.execute_input":"2024-06-13T08:06:30.785585Z","iopub.status.idle":"2024-06-13T08:06:31.082581Z","shell.execute_reply.started":"2024-06-13T08:06:30.785518Z","shell.execute_reply":"2024-06-13T08:06:31.081451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:31.084553Z","iopub.execute_input":"2024-06-13T08:06:31.085236Z","iopub.status.idle":"2024-06-13T08:06:31.433321Z","shell.execute_reply.started":"2024-06-13T08:06:31.085169Z","shell.execute_reply":"2024-06-13T08:06:31.432255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainGen.reset()\nvalGen.reset()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:31.435317Z","iopub.execute_input":"2024-06-13T08:06:31.436017Z","iopub.status.idle":"2024-06-13T08:06:31.441854Z","shell.execute_reply.started":"2024-06-13T08:06:31.435946Z","shell.execute_reply":"2024-06-13T08:06:31.440533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers[15:]:\n\tlayer.trainable = True\n\n# loop over the layers in the model and show which ones are trainable\n# or not\nfor layer in baseModel.layers:\n\tprint(\"{}: {}\".format(layer, layer.trainable))\n\n# for the changes to the model to take affect we need to recompile\n# the model, this time using SGD with a *very* small learning rate\nprint(\"[INFO] re-compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:31.444042Z","iopub.execute_input":"2024-06-13T08:06:31.444681Z","iopub.status.idle":"2024-06-13T08:06:31.513363Z","shell.execute_reply.started":"2024-06-13T08:06:31.444444Z","shell.execute_reply":"2024-06-13T08:06:31.512654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"H = model.fit_generator(\n\ttrainGen,\n\tsteps_per_epoch=totalTrain // BATCH_SIZE,\n\tvalidation_data=valGen,\n\tvalidation_steps=totalVal // BATCH_SIZE,\n\tepochs= epochsFineTune)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:06:31.51784Z","iopub.execute_input":"2024-06-13T08:06:31.518062Z","iopub.status.idle":"2024-06-13T08:19:31.497019Z","shell.execute_reply.started":"2024-06-13T08:06:31.51803Z","shell.execute_reply":"2024-06-13T08:19:31.495955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\nplot_training(H, epochsFineTune, UNFROZEN_PLOT_PATH)\n\n# serialize the model to disk\nprint(\"[INFO] serializing network...\")\nmodel.save(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:19:31.498729Z","iopub.execute_input":"2024-06-13T08:19:31.499044Z","iopub.status.idle":"2024-06-13T08:19:44.27019Z","shell.execute_reply.started":"2024-06-13T08:19:31.498994Z","shell.execute_reply":"2024-06-13T08:19:44.269088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:19:44.272307Z","iopub.execute_input":"2024-06-13T08:19:44.273161Z","iopub.status.idle":"2024-06-13T08:19:44.635343Z","shell.execute_reply.started":"2024-06-13T08:19:44.273078Z","shell.execute_reply":"2024-06-13T08:19:44.63449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:19:44.636859Z","iopub.execute_input":"2024-06-13T08:19:44.637184Z","iopub.status.idle":"2024-06-13T08:19:44.813262Z","shell.execute_reply.started":"2024-06-13T08:19:44.637126Z","shell.execute_reply":"2024-06-13T08:19:44.812486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:19:44.814704Z","iopub.execute_input":"2024-06-13T08:19:44.81501Z","iopub.status.idle":"2024-06-13T08:19:45.082933Z","shell.execute_reply.started":"2024-06-13T08:19:44.814954Z","shell.execute_reply":"2024-06-13T08:19:45.081922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from tensorflow.keras.applications import EfficientNetB0\nfrom efficientnet.tfkeras import EfficientNetB0\n\nfrom tensorflow.keras.layers import Input, Dense, Flatten, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import SGD\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\n\n# Define paths and other constants\ntrainPath = 'training'\nvalPath = 'validation'\ntestPath = 'evaluation'\nBATCH_SIZE = 32\ntrainEpochs = 10\n\n# Load EfficientNetB0 model\nbaseModel = EfficientNetB0(weights=\"imagenet\", include_top=False, input_tensor=Input(shape=(224, 224, 3)))\n\n# Construct the head of the model\nheadModel = baseModel.output\nheadModel = Flatten(name=\"flatten\")(headModel)\nheadModel = Dense(512, activation=\"relu\")(headModel)\nheadModel = Dropout(0.5)(headModel)\nheadModel = Dense(len(CLASSES), activation=\"softmax\")(headModel)\n\n# Combine the base model and head model\nmodel = Model(inputs=baseModel.input, outputs=headModel)\n\n# Freeze the layers in the base model\nfor layer in baseModel.layers:\n    layer.trainable = False\n\n# Compile the model\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt, metrics=[\"accuracy\"])\n\n# Print summary of the model\nmodel.summary()\n\n# Data generators\ntrain_datagen = ImageDataGenerator(rescale=1. / 255)\nval_datagen = ImageDataGenerator(rescale=1. / 255)\ntest_datagen = ImageDataGenerator(rescale=1. / 255)\n\ntrainGen = train_datagen.flow_from_directory(\n    trainPath,\n    target_size=(224, 224),\n    batch_size=BATCH_SIZE,\n    class_mode='categorical')\n\nvalGen = val_datagen.flow_from_directory(\n    valPath,\n    target_size=(224, 224),\n    batch_size=BATCH_SIZE,\n    class_mode='categorical')\n\ntestGen = test_datagen.flow_from_directory(\n    testPath,\n    target_size=(224, 224),\n    batch_size=BATCH_SIZE,\n    class_mode='categorical')\n\n# Train the head of the network\nH = model.fit(\n    trainGen,\n    steps_per_epoch=trainGen.samples // BATCH_SIZE,\n    validation_data=valGen,\n    validation_steps=valGen.samples // BATCH_SIZE,\n    epochs=trainEpochs)\n\n# Evaluate the model on test data\nprint(\"[INFO] evaluating model...\")\ntestGen.reset()\npredIdxs = model.predict(testGen, steps=(testGen.samples // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs, target_names=testGen.class_indices.keys()))\n\n# Plot training history\nplt.figure(figsize=(10, 6))\nplt.plot(np.arange(0, trainEpochs), H.history[\"loss\"], label=\"train_loss\")\nplt.plot(np.arange(0, trainEpochs), H.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(np.arange(0, trainEpochs), H.history[\"accuracy\"], label=\"train_acc\")\nplt.plot(np.arange(0, trainEpochs), H.history[\"val_accuracy\"], label=\"val_acc\")\nplt.title(\"Training Loss and Accuracy\")\nplt.xlabel(\"Epoch #\")\nplt.ylabel(\"Loss/Accuracy\")\nplt.legend(loc=\"lower left\")\nplt.savefig('efficientnet_training_plot.png')\n\n# Compute and plot confusion matrix\ntestGen.reset()\ny_pred_probs = model.predict(testGen, steps=(testGen.samples // BATCH_SIZE) + 1)\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\ny_true = testGen.classes\ncm = confusion_matrix(y_true, y_pred_labels)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted Label\")\nplt.ylabel(\"True Label\")\nplt.savefig('efficientnet_confusion_matrix.png')\nplt.show()\n\n# Save the model\nmodel.save('efficientnet_model.h5')\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:21.685777Z","iopub.execute_input":"2024-06-13T08:20:21.686114Z","iopub.status.idle":"2024-06-13T08:20:26.533437Z","shell.execute_reply.started":"2024-06-13T08:20:21.686064Z","shell.execute_reply":"2024-06-13T08:20:26.530311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install efficientnet\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:05.186328Z","iopub.execute_input":"2024-06-13T08:20:05.186702Z","iopub.status.idle":"2024-06-13T08:20:11.995166Z","shell.execute_reply.started":"2024-06-13T08:20:05.186641Z","shell.execute_reply":"2024-06-13T08:20:11.99404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from tensorflow.keras.applications import EfficientNetB0\nfrom efficientnet.tfkeras import EfficientNetB0\nfrom tensorflow.keras.layers import Input, Dense, Flatten, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import SGD\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Input tensor shape\ninput_shape = (224, 224, 3)\n\n# Load the EfficientNetB0 model pre-trained on ImageNet\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_tensor=Input(shape=input_shape))\n\n# Freeze the base model layers\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# Add custom head on top of the base model\nhead_model = base_model.output\nhead_model = Flatten(name=\"flatten\")(head_model)\nhead_model = Dense(512, activation=\"relu\")(head_model)\nhead_model = Dropout(0.5)(head_model)\nhead_model = Dense(len(CLASSES), activation=\"softmax\")(head_model)\n\n# Combine base model and custom head\nmodel = Model(inputs=base_model.input, outputs=head_model)\n\n# Compile the model\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt, metrics=[\"accuracy\"])\n\n# Print model summary\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:41.62384Z","iopub.execute_input":"2024-06-13T08:20:41.624147Z","iopub.status.idle":"2024-06-13T08:20:44.642256Z","shell.execute_reply.started":"2024-06-13T08:20:41.624102Z","shell.execute_reply":"2024-06-13T08:20:44.641241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Assuming testGen and model are defined\ntestGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\ny_true = testGen.classes\n\n# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5, 5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n\n# Generate and print classification report\nprint(classification_report(y_true, y_pred_labels, target_names=testGen.class_indices.keys()))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:53.213781Z","iopub.execute_input":"2024-06-13T08:20:53.214092Z","iopub.status.idle":"2024-06-13T08:21:07.912226Z","shell.execute_reply.started":"2024-06-13T08:20:53.21405Z","shell.execute_reply":"2024-06-13T08:21:07.911133Z"},"trusted":true},"execution_count":null,"outputs":[]}]}