{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":18147,"sourceType":"datasetVersion","datasetId":13405},{"sourceId":842050,"sourceType":"datasetVersion","datasetId":444558},{"sourceId":893807,"sourceType":"datasetVersion","datasetId":451078}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-24T13:07:38.738750Z","iopub.execute_input":"2024-02-24T13:07:38.739064Z","iopub.status.idle":"2024-02-24T13:07:39.283399Z","shell.execute_reply.started":"2024-02-24T13:07:38.739018Z","shell.execute_reply":"2024-02-24T13:07:39.282743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install facenet-pytorch","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:39.285278Z","iopub.execute_input":"2024-02-24T13:07:39.285534Z","iopub.status.idle":"2024-02-24T13:07:47.455658Z","shell.execute_reply.started":"2024-02-24T13:07:39.285483Z","shell.execute_reply":"2024-02-24T13:07:47.454825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\nimport numpy as np\nimport pandas as pd\nimport matplotlib\nimport seaborn as sns\nimport torch\nimport matplotlib.pyplot as plt\n# from tqdm import tqdm_notebook\n%matplotlib inline \n# from google.colab.patches import cv2_imshow\nfrom IPython.display import HTML #imports to play videos\nfrom base64 import b64encode \nimport cv2 as cv\nfrom skimage.measure import compare_ssim\nimport glob\nimport time\nfrom PIL import Image\nfrom facenet_pytorch import MTCNN, InceptionResnetV1, extract_face\nfrom tqdm import tqdm\n\nimport math\nimport pickle\nfrom functools import partial\nfrom collections import defaultdict\n\nfrom PIL import Image\nfrom glob import glob\n\nimport cv2\nimport skimage.measure\nimport albumentations as A\nfrom tqdm.notebook import tqdm \nfrom albumentations.pytorch import ToTensor \n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\nfrom torchvision.models.video import mc3_18, r2plus1d_18\n\nfrom facenet_pytorch import MTCNN","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:47.457579Z","iopub.execute_input":"2024-02-24T13:07:47.457931Z","iopub.status.idle":"2024-02-24T13:07:51.029136Z","shell.execute_reply.started":"2024-02-24T13:07:47.457865Z","shell.execute_reply":"2024-02-24T13:07:51.028201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = \"../input/deepfake-detection-challenge\" \nTRAIN_SAMPLE_FOLDER = \"train_sample_videos\"\nTEST_FOLDER = \"test_videos\"","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.030648Z","iopub.execute_input":"2024-02-24T13:07:51.030963Z","iopub.status.idle":"2024-02-24T13:07:51.035235Z","shell.execute_reply.started":"2024-02-24T13:07:51.030902Z","shell.execute_reply":"2024-02-24T13:07:51.034298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '../input/haarcascades'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")    ","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.038951Z","iopub.execute_input":"2024-02-24T13:07:51.039283Z","iopub.status.idle":"2024-02-24T13:07:51.051384Z","shell.execute_reply.started":"2024-02-24T13:07:51.039213Z","shell.execute_reply":"2024-02-24T13:07:51.050471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\") ","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.053909Z","iopub.execute_input":"2024-02-24T13:07:51.054197Z","iopub.status.idle":"2024-02-24T13:07:51.110518Z","shell.execute_reply.started":"2024-02-24T13:07:51.054146Z","shell.execute_reply":"2024-02-24T13:07:51.109645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.111560Z","iopub.execute_input":"2024-02-24T13:07:51.111814Z","iopub.status.idle":"2024-02-24T13:07:51.177570Z","shell.execute_reply.started":"2024-02-24T13:07:51.111762Z","shell.execute_reply":"2024-02-24T13:07:51.176861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")\n#reading the json file\ndef get_meta_from_json(path):\n    df = pd.read_json(os.path.join(DATA_FOLDER, path, json_file))\n    df = df.T\n    return df\n\nmeta_train_df = get_meta_from_json(TRAIN_SAMPLE_FOLDER)\nmeta_train_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.179144Z","iopub.execute_input":"2024-02-24T13:07:51.179460Z","iopub.status.idle":"2024-02-24T13:07:51.590840Z","shell.execute_reply.started":"2024-02-24T13:07:51.179402Z","shell.execute_reply":"2024-02-24T13:07:51.590104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.592131Z","iopub.execute_input":"2024-02-24T13:07:51.592433Z","iopub.status.idle":"2024-02-24T13:07:51.599378Z","shell.execute_reply.started":"2024-02-24T13:07:51.592382Z","shell.execute_reply":"2024-02-24T13:07:51.598695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.600748Z","iopub.execute_input":"2024-02-24T13:07:51.601018Z","iopub.status.idle":"2024-02-24T13:07:51.648795Z","shell.execute_reply.started":"2024-02-24T13:07:51.600956Z","shell.execute_reply":"2024-02-24T13:07:51.648056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df.loc[meta_train_df.label == 'REAL'])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.650183Z","iopub.execute_input":"2024-02-24T13:07:51.650437Z","iopub.status.idle":"2024-02-24T13:07:51.666975Z","shell.execute_reply.started":"2024-02-24T13:07:51.650387Z","shell.execute_reply":"2024-02-24T13:07:51.666145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Totals']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique() #collect all unique instances\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.668458Z","iopub.execute_input":"2024-02-24T13:07:51.668886Z","iopub.status.idle":"2024-02-24T13:07:51.675509Z","shell.execute_reply.started":"2024-02-24T13:07:51.668716Z","shell.execute_reply":"2024-02-24T13:07:51.674845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.676606Z","iopub.execute_input":"2024-02-24T13:07:51.676844Z","iopub.status.idle":"2024-02-24T13:07:51.696017Z","shell.execute_reply.started":"2024-02-24T13:07:51.676803Z","shell.execute_reply":"2024-02-24T13:07:51.695372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_frequent_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    items = []\n    vals = []\n    for col in data.columns:\n        itm = data[col].value_counts().index[0]\n        val = data[col].value_counts().values[0]\n        items.append(itm)\n        vals.append(val)\n    tt['Most frequent item'] = items\n    tt['Frequence'] = vals\n    tt['Percent from total'] = np.round(vals / total * 100, 3)\n    return(np.transpose(tt))\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.697218Z","iopub.execute_input":"2024-02-24T13:07:51.697442Z","iopub.status.idle":"2024-02-24T13:07:51.705222Z","shell.execute_reply.started":"2024-02-24T13:07:51.697404Z","shell.execute_reply":"2024-02-24T13:07:51.704377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.706539Z","iopub.execute_input":"2024-02-24T13:07:51.706885Z","iopub.status.idle":"2024-02-24T13:07:51.736878Z","shell.execute_reply.started":"2024-02-24T13:07:51.706835Z","shell.execute_reply":"2024-02-24T13:07:51.736040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df.loc[meta_train_df.label == 'FAKE'])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.738621Z","iopub.execute_input":"2024-02-24T13:07:51.738993Z","iopub.status.idle":"2024-02-24T13:07:51.831990Z","shell.execute_reply.started":"2024-02-24T13:07:51.738931Z","shell.execute_reply":"2024-02-24T13:07:51.831266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n  '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n  '''  \n  f, ax = plt.subplots(1,1, figsize=(4*size,4))\n  total = float(len(df))\n  g =  sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n  g.set_title(\"Number and percentage of {}\".format(title)) \n  if(size > 2):\n    plt.xticks(rotation=90, size=8)\n  for p in ax.patches:\n     height = p.get_height()\n     ax.text(p.get_x()+ p.get_width()/2.,height + 3,'{:1.2f}%'.format(100*height/total),ha=\"center\")\n\n  plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.833425Z","iopub.execute_input":"2024-02-24T13:07:51.833662Z","iopub.status.idle":"2024-02-24T13:07:51.843833Z","shell.execute_reply.started":"2024-02-24T13:07:51.833616Z","shell.execute_reply":"2024-02-24T13:07:51.843169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('split','split(train)',meta_train_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:51.845158Z","iopub.execute_input":"2024-02-24T13:07:51.845432Z","iopub.status.idle":"2024-02-24T13:07:52.075980Z","shell.execute_reply.started":"2024-02-24T13:07:51.845387Z","shell.execute_reply":"2024-02-24T13:07:52.074898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('label','label(train)',meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:52.077959Z","iopub.execute_input":"2024-02-24T13:07:52.078638Z","iopub.status.idle":"2024-02-24T13:07:52.309880Z","shell.execute_reply.started":"2024-02-24T13:07:52.078565Z","shell.execute_reply":"2024-02-24T13:07:52.308800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = np.array(list(meta_train_df.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:52.311866Z","iopub.execute_input":"2024-02-24T13:07:52.312568Z","iopub.status.idle":"2024-02-24T13:07:52.326071Z","shell.execute_reply.started":"2024-02-24T13:07:52.312494Z","shell.execute_reply":"2024-02-24T13:07:52.325062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='FAKE'].sample(3).index)\nfake_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:52.328072Z","iopub.execute_input":"2024-02-24T13:07:52.329044Z","iopub.status.idle":"2024-02-24T13:07:52.341288Z","shell.execute_reply.started":"2024-02-24T13:07:52.328953Z","shell.execute_reply":"2024-02-24T13:07:52.340170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_img = cv.VideoCapture(video_path)\n    ret, frame = capture_img.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:52.343345Z","iopub.execute_input":"2024-02-24T13:07:52.344021Z","iopub.status.idle":"2024-02-24T13:07:52.353378Z","shell.execute_reply.started":"2024-02-24T13:07:52.343939Z","shell.execute_reply":"2024-02-24T13:07:52.352366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video_file))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:52.355168Z","iopub.execute_input":"2024-02-24T13:07:52.355840Z","iopub.status.idle":"2024-02-24T13:07:53.715273Z","shell.execute_reply.started":"2024-02-24T13:07:52.355760Z","shell.execute_reply":"2024-02-24T13:07:53.714333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='REAL'].sample(3).index) #viewing the real videos\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:53.716985Z","iopub.execute_input":"2024-02-24T13:07:53.717475Z","iopub.status.idle":"2024-02-24T13:07:53.727271Z","shell.execute_reply.started":"2024-02-24T13:07:53.717263Z","shell.execute_reply":"2024-02-24T13:07:53.726323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video in real_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:53.728961Z","iopub.execute_input":"2024-02-24T13:07:53.729529Z","iopub.status.idle":"2024-02-24T13:07:54.930686Z","shell.execute_reply.started":"2024-02-24T13:07:53.729291Z","shell.execute_reply":"2024-02-24T13:07:54.929921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    #we only show images extracted from first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n      video_path = os.path.join(DATA_FOLDER, video_folder, video_file)\n      capture_img = cv.VideoCapture(video_path)\n      ret, frame = capture_img.read()\n      frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n      ax[i//3, i%3].imshow(frame)\n      ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n      ax[i//3, i%3].axis('on')","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:54.932097Z","iopub.execute_input":"2024-02-24T13:07:54.932533Z","iopub.status.idle":"2024-02-24T13:07:54.947972Z","shell.execute_reply.started":"2024-02-24T13:07:54.932484Z","shell.execute_reply":"2024-02-24T13:07:54.947166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='meawmsgiti.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:54.949375Z","iopub.execute_input":"2024-02-24T13:07:54.949659Z","iopub.status.idle":"2024-02-24T13:07:56.651688Z","shell.execute_reply.started":"2024-02-24T13:07:54.949607Z","shell.execute_reply":"2024-02-24T13:07:56.650911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvf /kaggle/input/ffmpeg-static-build/ffmpeg-git-amd64-static.tar.xz","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:07:56.653422Z","iopub.execute_input":"2024-02-24T13:07:56.653743Z","iopub.status.idle":"2024-02-24T13:08:00.956346Z","shell.execute_reply.started":"2024-02-24T13:07:56.653690Z","shell.execute_reply":"2024-02-24T13:08:00.955480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob, shutil\nimport timeit, os, gc\nimport subprocess as sp\nfrom tqdm import tqdm\nfrom collections import defaultdict\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nimport json\nfrom IPython.display import HTML\nfrom base64 import b64encode\nimport cv2\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:00.958246Z","iopub.execute_input":"2024-02-24T13:08:00.958605Z","iopub.status.idle":"2024-02-24T13:08:00.968741Z","shell.execute_reply.started":"2024-02-24T13:08:00.958541Z","shell.execute_reply":"2024-02-24T13:08:00.967882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOME = \"./\"\nFFMPEG = \"/kaggle/working/ffmpeg-git-20191209-amd64-static\"\nFFMPEG_PATH = FFMPEG\nDATA_FOLDER = \"/kaggle/input/deepfake-detection-challenge\"\nTMP_FOLDER = HOME\nDATA_FOLDER_TRAIN = DATA_FOLDER\nVIDEOS_FOLDER_TRAIN = DATA_FOLDER_TRAIN + \"/train_sample_videos\"\nIMAGES_FOLDER_TRAIN = TMP_FOLDER + \"/images\"\nAUDIOS_FOLDER_TRAIN = TMP_FOLDER + \"/audios\"\nEXTRACT_META = True # False\nEXTRACT_CONTENT = True # False\nEXTRACT_FACES = True # False\nFRAME_RATE = 0.5 # Frame per\nprint(FFMPEG)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:00.970147Z","iopub.execute_input":"2024-02-24T13:08:00.970417Z","iopub.status.idle":"2024-02-24T13:08:00.985920Z","shell.execute_reply.started":"2024-02-24T13:08:00.970368Z","shell.execute_reply":"2024-02-24T13:08:00.985104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_command(*popenargs, **kwargs):\n    closeNULL = 0\n    try:\n        from subprocess import DEVNULL\n        closeNULL = 0\n    except ImportError:\n        import os\n        DEVNULL = open(os.devnull, 'wb')\n        closeNULL = 1\n\n    process = sp.Popen(stdout=sp.PIPE, stderr=DEVNULL, *popenargs, **kwargs)\n    output, unused_err = process.communicate()\n    retcode = process.poll()\n\n    if closeNULL:\n        DEVNULL.close()\n\n    if retcode:\n        cmd = kwargs.get(\"args\")\n        if cmd is None:\n            cmd = popenargs[0]\n        error = sp.CalledProcessError(retcode, cmd)\n        error.output = output\n        raise error\n    return output\n\ndef ffprobe(filename, options = [\"-show_error\", \"-show_format\", \"-show_streams\", \"-show_programs\", \"-show_chapters\", \"-show_private_data\"]):\n    ret = {}\n    command = [FFMPEG_PATH + \"/ffprobe\", \"-v\", \"error\", *options, \"-print_format\", \"json\", filename]\n    ret = run_command(command)\n    if ret:\n        ret = json.loads(ret)\n    return ret\n\n# ffmpeg -i input.mov -r 0.25 output_%04d.png\ndef ffextract_frames(filename, output_folder, rate = 0.25):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-r\", str(rate), \"-y\", output_folder + \"/output_%04d.png\"]\n    ret = run_command(command)\n    return ret\n\n# ffmpeg -i input-video.mp4 output-audio.mp3\ndef ffextract_audio(filename, output_path):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-vn\", \"-ac\", \"1\", \"-acodec\", \"copy\", \"-y\", output_path]\n    ret = run_command(command)\n    return ret","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:00.987507Z","iopub.execute_input":"2024-02-24T13:08:00.987792Z","iopub.status.idle":"2024-02-24T13:08:01.002367Z","shell.execute_reply.started":"2024-02-24T13:08:00.987733Z","shell.execute_reply":"2024-02-24T13:08:01.001598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTRACT_META == True:\n    results = []\n    subfolder = VIDEOS_FOLDER_TRAIN\n    filepaths = glob.glob(subfolder + \"/*.mp4\")\n    for filepath in tqdm(filepaths):\n        js = ffprobe(filepath)\n#        print(js)\n        if js:\n            results.append(\n                (js.get(\"format\", {}).get(\"filename\")[len(subfolder) + 1:],\n                js.get(\"format\", {}).get(\"format_long_name\"),\n                # Video \n                js.get(\"streams\", [{}, {}])[0].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"height\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"width\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"start_time\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"avg_frame_rate\"),\n                 # Audio\n                js.get(\"streams\", [{}, {}])[1].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"channels\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"sample_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"start_time\")),\n            )\n\n    meta_pd = pd.DataFrame(results, columns=[\"filename\", \"format\", \"video_codec_name\", \"video_height\", \"video_width\",\n                                            \"video_nb_frames\", \"video_bit_rate\", \"video_duration\", \"video_start_time\",\"video_fps\",\n                                            \"audio_codec_name\", \"audio_channels\", \"audio_sample_rate\", \"audio_nb_frames\",\n                                            \"audio_bit_rate\", \"audio_duration\", \"audio_start_time\"])\n    meta_pd[\"video_fps\"] = meta_pd[\"video_fps\"].apply(lambda x: float(x.split(\"/\")[0])/float(x.split(\"/\")[1]) if len(x.split(\"/\")) == 2 else None)\n    meta_pd[\"video_duration\"] = meta_pd[\"video_duration\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"video_start_time\"] = meta_pd[\"video_start_time\"].astype(np.float32)\n    meta_pd[\"video_nb_frames\"] = meta_pd[\"video_nb_frames\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_sample_rate\"] = meta_pd[\"audio_sample_rate\"].astype(np.float32)\n    meta_pd[\"audio_nb_frames\"] = meta_pd[\"audio_nb_frames\"].astype(np.float32)\n    meta_pd[\"audio_bit_rate\"] = meta_pd[\"audio_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_duration\"] = meta_pd[\"audio_duration\"].astype(np.float32)\n    meta_pd[\"audio_start_time\"] = meta_pd[\"audio_start_time\"].astype(np.float32)\n    meta_pd.to_pickle(HOME + \"videos_meta.pkl\")\nelse:\n    meta_pd = pd.read_pickle(HOME + \"videos_meta.pkl\")\nmeta_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:01.011697Z","iopub.execute_input":"2024-02-24T13:08:01.011926Z","iopub.status.idle":"2024-02-24T13:08:19.191199Z","shell.execute_reply.started":"2024-02-24T13:08:01.011886Z","shell.execute_reply":"2024-02-24T13:08:19.190214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,6, figsize=(22, 3))\nd = sns.distplot(meta_pd[\"video_fps\"], ax=ax[0])\nd = sns.distplot(meta_pd[\"video_duration\"], ax=ax[1])\nd = sns.distplot(meta_pd[\"video_width\"], ax=ax[2])\nd = sns.distplot(meta_pd[\"video_height\"], ax=ax[3])\nd = sns.distplot(meta_pd[\"video_nb_frames\"], ax=ax[4])\nd = sns.distplot(meta_pd[\"video_bit_rate\"], ax=ax[5])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:19.193115Z","iopub.execute_input":"2024-02-24T13:08:19.193447Z","iopub.status.idle":"2024-02-24T13:08:21.065789Z","shell.execute_reply.started":"2024-02-24T13:08:19.193385Z","shell.execute_reply":"2024-02-24T13:08:21.064374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:21.067881Z","iopub.execute_input":"2024-02-24T13:08:21.068461Z","iopub.status.idle":"2024-02-24T13:08:21.266432Z","shell.execute_reply.started":"2024-02-24T13:08:21.068236Z","shell.execute_reply":"2024-02-24T13:08:21.265431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:21.267873Z","iopub.execute_input":"2024-02-24T13:08:21.268238Z","iopub.status.idle":"2024-02-24T13:08:21.444598Z","shell.execute_reply.started":"2024-02-24T13:08:21.268169Z","shell.execute_reply":"2024-02-24T13:08:21.443750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.merge(train_pd, meta_pd[[\"filename\", \"video_height\", \"video_width\", \"video_nb_frames\", \"video_bit_rate\", \"audio_nb_frames\"]], on=\"filename\", how=\"left\")\ntrain_pd[\"count\"] = train_pd.groupby([\"original\"])[\"original\"].transform('count')\n# train_pd.to_pickle(HOME + \"train_meta.pkl\")\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:21.445842Z","iopub.execute_input":"2024-02-24T13:08:21.446097Z","iopub.status.idle":"2024-02-24T13:08:21.479495Z","shell.execute_reply.started":"2024-02-24T13:08:21.446052Z","shell.execute_reply":"2024-02-24T13:08:21.478814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUDIO_FORMAT = \"aac\" # \"wav\"\nvideos_folder = VIDEOS_FOLDER_TRAIN\nimages_folder_path = IMAGES_FOLDER_TRAIN\naudios_folder_path = AUDIOS_FOLDER_TRAIN\nif EXTRACT_CONTENT == True:\n    # 1h20min for chunk#0 (11GB)\n    # Extract some images + audio track\n    for idx, row in tqdm(train_pd.iterrows(), total=meta_pd.shape[0]):\n        try:\n            video_path = videos_folder + \"/\" + row[\"filename\"]\n            images_path = images_folder_path + \"/\" + row[\"filename\"][:-4]\n            audio_path = audios_folder_path + \"/\" + row[\"filename\"][:-4]\n            # Extract images\n            if not os.path.exists(images_path): os.makedirs(images_path)\n            ret = ffextract_frames(video_path, images_path, rate = FRAME_RATE)\n            # Extract audio\n            if not os.path.exists(audio_path): os.makedirs(audio_path)\n            # ret = ffextract_audio(video_path, audio_path + \"/audio.\" + AUDIO_FORMAT)\n        except:\n            print(\"Cannot extract frames/audio for:\" + row[\"filename\"])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:08:21.480576Z","iopub.execute_input":"2024-02-24T13:08:21.480789Z","iopub.status.idle":"2024-02-24T13:20:11.208193Z","shell.execute_reply.started":"2024-02-24T13:08:21.480752Z","shell.execute_reply":"2024-02-24T13:20:11.207233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd.tail()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:11.210099Z","iopub.execute_input":"2024-02-24T13:20:11.210420Z","iopub.status.idle":"2024-02-24T13:20:11.229725Z","shell.execute_reply.started":"2024-02-24T13:20:11.210363Z","shell.execute_reply":"2024-02-24T13:20:11.229067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 21 # 27 # 21 # 19 # 12 # 6\nfake = train_pd[\"filename\"][idx]\nreal = train_pd[\"original\"][idx]\nvid_width = train_pd[\"video_width\"][idx]\nvid_real = open(VIDEOS_FOLDER_TRAIN + \"/\" + real, 'rb').read()\ndata_url_real = \"data:video/mp4;base64,\" + b64encode(vid_real).decode()\nvid_fake = open(VIDEOS_FOLDER_TRAIN + \"/\" + fake, 'rb').read()\ndata_url_fake = \"data:video/mp4;base64,\" + b64encode(vid_fake).decode()\nHTML(\"\"\"\n<div style='width: 100%%; display: table;'>\n    <div style='display: table-row'>\n        <div style='width: %dpx; display: table-cell;'><b>Real</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n        <div style='display: table-cell;'><b>Fake</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n    </div>\n</div>\n\"\"\" % ( int(vid_width/3.2) + 10, \n       real, int(vid_width/3.2), data_url_real, \n       fake, int(vid_width/3.2), data_url_fake))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:11.231167Z","iopub.execute_input":"2024-02-24T13:20:11.231451Z","iopub.status.idle":"2024-02-24T13:20:11.704970Z","shell.execute_reply.started":"2024-02-24T13:20:11.231400Z","shell.execute_reply":"2024-02-24T13:20:11.703370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + \"haarcascade_frontalface_default.xml\")\n\ndef detect_face_cv2(img):\n    # Move to grayscale\n    gray_img = cv2.cvtColor(img.copy(), cv2.COLOR_RGB2GRAY)\n    face_locations = []\n    face_rects = face_cascade.detectMultiScale(gray_img, scaleFactor=1.3, minNeighbors=5)     \n    for (x,y,w,h) in face_rects: \n        face_location = (x,y,w,h)\n        face_locations.append((face_location, 1.0))\n    return face_locations","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:11.707341Z","iopub.execute_input":"2024-02-24T13:20:11.707792Z","iopub.status.idle":"2024-02-24T13:20:11.767508Z","shell.execute_reply.started":"2024-02-24T13:20:11.707715Z","shell.execute_reply":"2024-02-24T13:20:11.766810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mtcnn","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:11.768729Z","iopub.execute_input":"2024-02-24T13:20:11.769021Z","iopub.status.idle":"2024-02-24T13:20:18.429153Z","shell.execute_reply.started":"2024-02-24T13:20:11.768960Z","shell.execute_reply":"2024-02-24T13:20:18.428029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mtcnn import MTCNN\ndetector = MTCNN()\n\ndef detect_face_mtcnn(img):\n    face_locations = []\n    items = detector.detect_faces(img)\n    for face in items:\n        face_location = tuple(face.get('box'))\n        face_confidence = float(face.get('confidence'))\n        face_locations.append((face_location, face_confidence))\n    return face_locations","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:18.430660Z","iopub.execute_input":"2024-02-24T13:20:18.430925Z","iopub.status.idle":"2024-02-24T13:20:29.272167Z","shell.execute_reply.started":"2024-02-24T13:20:18.430878Z","shell.execute_reply":"2024-02-24T13:20:29.271198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_faces(files, source, detector=detect_face_cv2):\n    results = []\n    # for idx, file in tqdm(enumerate(files), total=len(files)):\n    for idx, file in enumerate(files):\n        try:\n            img = cv2.cvtColor(cv2.imread(file, cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)\n            face_locations = detector(img)\n            results.append((source, file[file.find(\"output_\"):], face_locations, len(face_locations)))\n        except:\n            print(\"Cannot extract faces for image: %s\" % file)\n    return results","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:29.273618Z","iopub.execute_input":"2024-02-24T13:20:29.273941Z","iopub.status.idle":"2024-02-24T13:20:29.281688Z","shell.execute_reply.started":"2024-02-24T13:20:29.273881Z","shell.execute_reply":"2024-02-24T13:20:29.280825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = fake\ndump_folder = images_folder_path + \"/\" + file[:-4]\nfiles = glob.glob(dump_folder + \"/*\")\nDETECTORS = {\n    \"cv2\": detect_face_cv2,\n    \"mtcnn\": detect_face_mtcnn\n}\nfaces_pd = None\nfor key, value in DETECTORS.items():\n    tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n    if faces_pd is None:\n        faces_pd = tmp_pd\n    else:\n        faces_pd = pd.merge(faces_pd, tmp_pd, on=[\"filename\", \"image\"], how=\"left\")\nfaces_pd.head(12)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:29.283174Z","iopub.execute_input":"2024-02-24T13:20:29.283491Z","iopub.status.idle":"2024-02-24T13:20:39.165836Z","shell.execute_reply.started":"2024-02-24T13:20:29.283433Z","shell.execute_reply":"2024-02-24T13:20:39.165087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_faces_boxes(df, max_cols = 2, max_rows = 6, fsize=(24, 5), max_items=12):    \n    idx = 0    \n    for item_idx, item in df.iterrows():\n        img = cv2.cvtColor(cv2.imread(IMAGES_FOLDER_TRAIN + \"/\" + item[\"filename\"][:-4] +\"/\" + item[\"image\"], cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)    \n        face_img = img #.copy()\n        # grid subplots\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0: fig = plt.figure(figsize=fsize)\n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        # display image with boxes\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols, 0):\n            face_locations = item[c]\n            face_confidence = item[c]            \n            if len(face_locations) > 0:\n                for face_location in face_locations:        \n                    ((x,y,w,h), confidence) = face_location\n                    # face_img = face_img[y:y+h, x:x+w]\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255,i*255,0), 8)\n                    cv2.putText(face_img, '%.1f' % (confidence*100.0), (x+w, y+h), cv2.FONT_HERSHEY_SIMPLEX, 2.0, (255,i*255,0), 9, cv2.LINE_AA)\n                ax.imshow(face_img)\n            else:\n                ax.imshow(img)\n            ax.set_title(\"%s %s / %s - Faces: %d %s %s\" % (item[\"label\"] if \"label\" in df.columns else \"\", \n                                                           item[\"filename\"], item[\"image\"],\n                                                           item[\"faces_mtcnn\"] if \"faces_mtcnn\" in df.columns else len(face_locations),\n                                                           item[\"faces_mtcnn_median\"] if \"faces_mtcnn_median\" in df.columns else \"\",\n                                                           item[\"faces\"] if \"faces\" in df.columns else \"\"))\n        if (col == max_cols -1): plt.show()\n        idx = idx + 1\n        if (max_items > 0 and idx >=max_items): break","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:39.167202Z","iopub.execute_input":"2024-02-24T13:20:39.167431Z","iopub.status.idle":"2024-02-24T13:20:39.192551Z","shell.execute_reply.started":"2024-02-24T13:20:39.167389Z","shell.execute_reply":"2024-02-24T13:20:39.191740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(faces_pd)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:39.193962Z","iopub.execute_input":"2024-02-24T13:20:39.194265Z","iopub.status.idle":"2024-02-24T13:20:41.907932Z","shell.execute_reply.started":"2024-02-24T13:20:39.194214Z","shell.execute_reply":"2024-02-24T13:20:41.907216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_detector_on_video(videos_filename, verbose=False):\n    if verbose == True: \n        print(\"Starting with batch of %d videos\" % len(videos_filename))\n    tmp_faces_pd = None\n    for file in videos_filename:\n        # Find out dump folder with images\n        dump_folder = images_folder_path + \"/\" + file[:-4]\n        # List files\n        files = glob.glob(dump_folder + \"/*\")\n        DETECTORS = {\n            \"mtcnn\": detect_face_mtcnn\n        }\n        for key, value in DETECTORS.items():\n            tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n            if tmp_faces_pd is None:\n                tmp_faces_pd = tmp_pd\n            else:\n                tmp_faces_pd = pd.concat([tmp_faces_pd, tmp_pd], axis=0)\n    return tmp_faces_pd","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:41.909191Z","iopub.execute_input":"2024-02-24T13:20:41.909470Z","iopub.status.idle":"2024-02-24T13:20:41.917959Z","shell.execute_reply.started":"2024-02-24T13:20:41.909419Z","shell.execute_reply":"2024-02-24T13:20:41.917127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\ncpus = multiprocessing.cpu_count()\nif EXTRACT_FACES == True:\n    resultfutures = []\n    results = []\n    tasks = np.array_split(train_pd[\"filename\"].unique(), 20)\n    print(\"Tasks: %d\" % len(tasks))\n    with ThreadPoolExecutor(max_workers=cpus) as executor:\n        resultfutures = tqdm(executor.map(run_detector_on_video, tasks), total=len(tasks))\n    results = [x for x in resultfutures]\n    executor.shutdown()\n    # Gather results\n    all_faces_pd = None\n    for result in results:\n        if all_faces_pd is None:\n            all_faces_pd = result\n        else:\n            all_faces_pd = pd.concat([all_faces_pd, result], axis=0)\n    all_faces_pd = all_faces_pd.reset_index(drop=True)\n    all_faces_pd.to_pickle(HOME + \"faces.pkl\")\nelse:\n    all_faces_pd = pd.read_pickle(HOME + \"faces.pkl\")\nprint(all_faces_pd.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:20:41.919580Z","iopub.execute_input":"2024-02-24T13:20:41.919907Z","iopub.status.idle":"2024-02-24T13:46:05.966330Z","shell.execute_reply.started":"2024-02-24T13:20:41.919847Z","shell.execute_reply":"2024-02-24T13:46:05.965585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_faces_pd[\"faces_mtcnn_avg\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmean)\nall_faces_pd[\"faces_mtcnn_median\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmedian)\nall_faces_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:05.967754Z","iopub.execute_input":"2024-02-24T13:46:05.967990Z","iopub.status.idle":"2024-02-24T13:46:05.998264Z","shell.execute_reply.started":"2024-02-24T13:46:05.967949Z","shell.execute_reply":"2024-02-24T13:46:05.997579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_avg\"], kde=True, ax=ax[0])\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_median\"], kde=False, ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:05.999382Z","iopub.execute_input":"2024-02-24T13:46:05.999597Z","iopub.status.idle":"2024-02-24T13:46:06.807108Z","shell.execute_reply.started":"2024-02-24T13:46:05.999560Z","shell.execute_reply":"2024-02-24T13:46:06.805839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(all_faces_pd[all_faces_pd[\"faces_mtcnn\"] == 3], max_items=24)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:06.809679Z","iopub.execute_input":"2024-02-24T13:46:06.810095Z","iopub.status.idle":"2024-02-24T13:46:14.263632Z","shell.execute_reply.started":"2024-02-24T13:46:06.810029Z","shell.execute_reply":"2024-02-24T13:46:14.262772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_faces_pd = pd.merge(all_faces_pd, train_pd, on=\"filename\", how=\"left\")\nclean_faces_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:14.264893Z","iopub.execute_input":"2024-02-24T13:46:14.265187Z","iopub.status.idle":"2024-02-24T13:46:14.298236Z","shell.execute_reply.started":"2024-02-24T13:46:14.265136Z","shell.execute_reply":"2024-02-24T13:46:14.297474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def faces_max_item(boxes, idx1, idx2):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[idx1][idx2])[idx1][idx2]\n    return ret\n\ndef faces_max_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[1])[1]\n    return ret\n\ndef faces_min_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = min(boxes, key=lambda item: item[1])[1]\n    return ret\n\nclean_faces_pd[\"faces_max_width\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 2)) \nclean_faces_pd[\"faces_max_height\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 3))\nclean_faces_pd[\"faces_max_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_confidence(x))\nclean_faces_pd[\"faces_min_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_min_confidence(x))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:14.299612Z","iopub.execute_input":"2024-02-24T13:46:14.300097Z","iopub.status.idle":"2024-02-24T13:46:14.338393Z","shell.execute_reply.started":"2024-02-24T13:46:14.299860Z","shell.execute_reply":"2024-02-24T13:46:14.337654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Faces stats:\")\nprint(clean_faces_pd[[\"faces_max_width\", \"faces_max_height\", \"faces_min_conf\", \"faces_max_conf\"]].describe(percentiles=[0.01,0.05, 0.1,0.25,0.5,0.75,0.9,0.95,0.99]))\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_max_width\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_height\"], kde=True, ax=ax[1])\nplt.show()\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_min_conf\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_conf\"], kde=True, ax=ax[1])\nfig, ax = plt.subplots(figsize=(22, 3))\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_width\", y=\"faces_max_conf\", c=\"red\", ax=ax, label=\"faces_max_width\", alpha=0.5)\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_height\", y=\"faces_max_conf\", c=\"blue\", ax=d,  label=\"faces_max_height\", alpha=0.5)\nd = plt.legend(loc=\"upper right\")","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:14.339775Z","iopub.execute_input":"2024-02-24T13:46:14.340063Z","iopub.status.idle":"2024-02-24T13:46:15.892446Z","shell.execute_reply.started":"2024-02-24T13:46:14.340008Z","shell.execute_reply":"2024-02-24T13:46:15.891775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install imutils","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:15.893705Z","iopub.execute_input":"2024-02-24T13:46:15.894181Z","iopub.status.idle":"2024-02-24T13:46:23.963374Z","shell.execute_reply.started":"2024-02-24T13:46:15.894127Z","shell.execute_reply":"2024-02-24T13:46:23.962379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport shutil\nimport cv2\nimport pandas as pd\nimport matplotlib\nmatplotlib.use(\"Agg\")\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pickle\nfrom imutils import paths\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import VGG16\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dense\nfrom keras.layers import Input\nfrom keras.models import Model\nfrom keras.optimizers import SGD\nfrom sklearn.metrics import classification_report\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:23.965521Z","iopub.execute_input":"2024-02-24T13:46:23.965891Z","iopub.status.idle":"2024-02-24T13:46:24.323590Z","shell.execute_reply.started":"2024-02-24T13:46:23.965825Z","shell.execute_reply":"2024-02-24T13:46:24.322810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/working/finetuningkeras/dataset\"\n\n# define the names of the training, testing, and validation\n# directories\nTRAIN = \"training\"\nTEST = \"evaluation\"\nVAL = \"validation\"\n\nREAL = 'REAL'\nFAKE = 'FAKE'\n\n# initialize the list of class label names\nCLASSES = [\"FAKE\", \"REAL\"]\n\n\n# set the batch size when fine-tuning\nBATCH_SIZE = 32\n\ntrainEpochs = 10\nepochsFineTune = 10\nmaxVids = 5\n\n# set the path to the serialized model after training\nMODEL_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"Deepfake.model\"])\n\n# define the path to the output training history plots\nUNFROZEN_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"unfrozen.png\"])\nWARMUP_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"warmup.png\"])\n\nfile = '/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json'\nimg_path = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ndata_path = '/kaggle/working/finetuningkeras/real_fake'\ndir_fake_frames = '/kaggle/working/FAKE_frames'\ndir_real_frames = '/kaggle/working/REAL_frames'\ndir_output = '/kaggle/working/finetuningkeras/output'\n\ndir_data_path_real = os.path.join(data_path, REAL)\ndir_data_path_fake = os.path.join(data_path, FAKE)\n\ndir_train_real = os.path.join(BASE_PATH, TRAIN, REAL)\ndir_train_fake = os.path.join(BASE_PATH, TRAIN, FAKE)\ndir_valid_real = os.path.join(BASE_PATH, VAL, REAL)\ndir_valid_fake = os.path.join(BASE_PATH, VAL, FAKE)\ndir_test_real = os.path.join(BASE_PATH, TEST, REAL)\ndir_test_fake = os.path.join(BASE_PATH, TEST, FAKE)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:24.324970Z","iopub.execute_input":"2024-02-24T13:46:24.325334Z","iopub.status.idle":"2024-02-24T13:46:24.338528Z","shell.execute_reply.started":"2024-02-24T13:46:24.325264Z","shell.execute_reply":"2024-02-24T13:46:24.337732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = '/kaggle/working/finetuningkeras/real_fake/FAKE'\noutput_dir = '/kaggle/working/FAKE_frames/'\ndef explode_frames(input_dir, output_dir, maxN):\n\n    mp4_filenames = [f for f in os.listdir(input_dir) if f.endswith('.mp4')]\n    n = 0\n    \n    for mp4fn in mp4_filenames:\n        \n        if(n < maxN):\n            n += 1 \n            mp4fp = os.path.join(input_dir, mp4fn)\n            cam = cv2.VideoCapture(mp4fp) \n            if(cam.isOpened()):\n                print('Processing file #'+ str(n) + ' (' + mp4fn + ')...')\n            else: \n                print('Problem opening file #'+ str(n) + ' (' + mp4fn + ')...')\n                continue \n            \n            nframe = 0\n            while(True): #continue until ret = False then break\n                nframe += 1\n                ret,frame = cam.read()\n                \n                if ret: \n                    # if video is still left continue creating images \n                    out_filename = os.path.splitext(mp4fn)[0]+  '_frame' + str(nframe) + '.jpg'\n                    out_filepath =  os.path.join(output_dir, out_filename)\n                    \n                    # writing the extracted images \n                    cv2.imwrite(out_filepath, frame) \n                else: \n                    break\n\n            # Release all space and windows once done\n            print(' - created ' + str(nframe-1) + ' images') # -1 bc count incremented before exit\n            cam.release() \n            cv2.destroyAllWindows()\n            \n        else: \n            break\n            \n\"\"\"\nDistribute files/images from a source directory into training, validation, and testing directories. \nsrc_dir = source/input directory\ntrain_dir, val_dir, test_dir = target training/validation/testing directory\nvalperc = fraction of dataset to use for validation (0-1)\ntestperc = fraction of dataset to use for testing (0-1)\n\"\"\"\n            \ndef trainvaltest_split(src_dir, train_dir, val_dir, test_dir, valperc = 0.15, testperc = 0.15):\n    \n    filenames = os.listdir(src_dir) #get all filenames in random order\n    np.random.shuffle(filenames)\n    \n    n = len(filenames)\n    split1 = int(n*(1 - (valperc + testperc)))\n    split2 = int(n*(1 - (testperc)))\n    \n    fn_train, fn_val, fn_test = np.split(np.array(filenames), [split1, split2])\n    \n    fn_lists = [fn_train, fn_val, fn_test]\n    targetdirs = [train_dir, val_dir, test_dir]\n    \n    print('Total images: ', n)\n    print('Training: ', len(fn_train))\n    print('Validation: ', len(fn_val))\n    print('Testing: ', len(fn_test))\n    \n    all_fp = [os.path.join(src_dir, fn) for fn in filenames]\n    \n    #move files\n    for i, fn_list in enumerate(fn_lists):\n        for fn in fn_list: \n            target_dir = targetdirs[i]\n            fp_from = os.path.join(src_dir, fn)\n            fp_to = os.path.join(target_dir, fn)\n            \n            shutil.move(fp_from, fp_to)\n\n            \n\"\"\"\nConstruct a plot that plots and saves the training history\n\"\"\"           \ndef plot_training(H, N, plotPath):\n\tplt.style.use(\"ggplot\")\n\tplt.figure()\n\tplt.plot(np.arange(0, N), H.history[\"loss\"], label=\"train_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"val_loss\"], label=\"val_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"accuracy\"], label=\"train_acc\")\n\tplt.plot(np.arange(0, N), H.history[\"val_accuracy\"], label=\"val_acc\")\n\tplt.title(\"Training Loss and Accuracy\")\n\tplt.xlabel(\"Epoch #\")\n\tplt.ylabel(\"Loss/Accuracy\")\n\tplt.legend(loc=\"lower left\")\n\tplt.savefig(plotPath)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:24.340123Z","iopub.execute_input":"2024-02-24T13:46:24.340474Z","iopub.status.idle":"2024-02-24T13:46:24.368792Z","shell.execute_reply.started":"2024-02-24T13:46:24.340416Z","shell.execute_reply":"2024-02-24T13:46:24.368129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(dir_train_real, exist_ok = True)\nos.makedirs(dir_train_fake, exist_ok = True)\nos.makedirs(dir_valid_real, exist_ok = True)\nos.makedirs(dir_valid_fake, exist_ok = True)\nos.makedirs(dir_test_real, exist_ok = True)\nos.makedirs(dir_test_fake, exist_ok = True)\n\nos.makedirs(dir_data_path_real, exist_ok = True)\nos.makedirs(dir_data_path_fake, exist_ok = True)\nos.makedirs(dir_fake_frames, exist_ok = True) \nos.makedirs(dir_real_frames, exist_ok = True) \nos.makedirs(dir_output, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:24.370396Z","iopub.execute_input":"2024-02-24T13:46:24.370725Z","iopub.status.idle":"2024-02-24T13:46:24.384785Z","shell.execute_reply.started":"2024-02-24T13:46:24.370668Z","shell.execute_reply":"2024-02-24T13:46:24.383992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_json(file)\ndf = df.T\n\n# %% [code]\nlabel = df[['label']]","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:24.386083Z","iopub.execute_input":"2024-02-24T13:46:24.386568Z","iopub.status.idle":"2024-02-24T13:46:24.560800Z","shell.execute_reply.started":"2024-02-24T13:46:24.386353Z","shell.execute_reply":"2024-02-24T13:46:24.559856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fn, row in label.iterrows():\n    src = os.path.join(img_path, fn)\n    dest = os.path.join(data_path, row['label'], fn)\n    shutil.copy(src, dest)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:24.562455Z","iopub.execute_input":"2024-02-24T13:46:24.562801Z","iopub.status.idle":"2024-02-24T13:46:27.718018Z","shell.execute_reply.started":"2024-02-24T13:46:24.562740Z","shell.execute_reply":"2024-02-24T13:46:27.717197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_fake, dir_fake_frames, maxN= maxVids)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:46:27.719300Z","iopub.execute_input":"2024-02-24T13:46:27.719548Z","iopub.status.idle":"2024-02-24T13:47:17.861371Z","shell.execute_reply.started":"2024-02-24T13:46:27.719507Z","shell.execute_reply":"2024-02-24T13:47:17.860580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_real, dir_real_frames, maxN= maxVids)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:47:17.862821Z","iopub.execute_input":"2024-02-24T13:47:17.863163Z","iopub.status.idle":"2024-02-24T13:48:10.913066Z","shell.execute_reply.started":"2024-02-24T13:47:17.863102Z","shell.execute_reply":"2024-02-24T13:48:10.912226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_fake_frames,\n                   train_dir = dir_train_fake, \n                   val_dir = dir_valid_fake, \n                   test_dir = dir_test_fake)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:10.914427Z","iopub.execute_input":"2024-02-24T13:48:10.914670Z","iopub.status.idle":"2024-02-24T13:48:10.974354Z","shell.execute_reply.started":"2024-02-24T13:48:10.914631Z","shell.execute_reply":"2024-02-24T13:48:10.973592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_real_frames,\n                   train_dir = dir_train_real, \n                   val_dir = dir_valid_real, \n                   test_dir = dir_test_real)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:10.975865Z","iopub.execute_input":"2024-02-24T13:48:10.976203Z","iopub.status.idle":"2024-02-24T13:48:11.036018Z","shell.execute_reply.started":"2024-02-24T13:48:10.976144Z","shell.execute_reply":"2024-02-24T13:48:11.035223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrainAug = ImageDataGenerator(\n\trotation_range=30,\n\tzoom_range=0.15,\n\twidth_shift_range=0.2,\n\theight_shift_range=0.2,\n\tshear_range=0.15,\n\thorizontal_flip=True,\n\tfill_mode=\"nearest\")","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:11.037401Z","iopub.execute_input":"2024-02-24T13:48:11.037733Z","iopub.status.idle":"2024-02-24T13:48:11.042526Z","shell.execute_reply.started":"2024-02-24T13:48:11.037669Z","shell.execute_reply":"2024-02-24T13:48:11.041802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valAug = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:11.043826Z","iopub.execute_input":"2024-02-24T13:48:11.044199Z","iopub.status.idle":"2024-02-24T13:48:11.051923Z","shell.execute_reply.started":"2024-02-24T13:48:11.044147Z","shell.execute_reply":"2024-02-24T13:48:11.051123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = np.array([123.68, 116.779, 103.939], dtype=\"float32\")\ntrainAug.mean = mean\nvalAug.mean = mean","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:11.053135Z","iopub.execute_input":"2024-02-24T13:48:11.053472Z","iopub.status.idle":"2024-02-24T13:48:11.061189Z","shell.execute_reply.started":"2024-02-24T13:48:11.053413Z","shell.execute_reply":"2024-02-24T13:48:11.060356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainPath = os.path.join(BASE_PATH, TRAIN)\ntrainGen = trainAug.flow_from_directory(\n\ttrainPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=True,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the validation generator\nvalPath = os.path.join(BASE_PATH, VAL)\nvalGen = valAug.flow_from_directory(\n\tvalPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the testing generator\ntestPath = os.path.join(BASE_PATH, TEST)\ntestGen = valAug.flow_from_directory(\n\ttestPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:11.062590Z","iopub.execute_input":"2024-02-24T13:48:11.062903Z","iopub.status.idle":"2024-02-24T13:48:11.385651Z","shell.execute_reply.started":"2024-02-24T13:48:11.062848Z","shell.execute_reply":"2024-02-24T13:48:11.384953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseModel = VGG16(weights=\"imagenet\", include_top=False,\n\tinput_tensor=Input(shape=(224, 224, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:11.387244Z","iopub.execute_input":"2024-02-24T13:48:11.387583Z","iopub.status.idle":"2024-02-24T13:48:12.379801Z","shell.execute_reply.started":"2024-02-24T13:48:11.387523Z","shell.execute_reply":"2024-02-24T13:48:12.379135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headModel = baseModel.output\nheadModel = Flatten(name=\"flatten\")(headModel)\nheadModel = Dense(512, activation=\"relu\")(headModel)\nheadModel = Dropout(0.5)(headModel)\nheadModel = Dense(len(CLASSES), activation=\"softmax\")(headModel)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:12.381010Z","iopub.execute_input":"2024-02-24T13:48:12.381234Z","iopub.status.idle":"2024-02-24T13:48:12.423701Z","shell.execute_reply.started":"2024-02-24T13:48:12.381195Z","shell.execute_reply":"2024-02-24T13:48:12.423007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=baseModel.input, outputs=headModel)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:12.425161Z","iopub.execute_input":"2024-02-24T13:48:12.425487Z","iopub.status.idle":"2024-02-24T13:48:12.431017Z","shell.execute_reply.started":"2024-02-24T13:48:12.425428Z","shell.execute_reply":"2024-02-24T13:48:12.430275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers:\n\tlayer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:12.432586Z","iopub.execute_input":"2024-02-24T13:48:12.432906Z","iopub.status.idle":"2024-02-24T13:48:12.446843Z","shell.execute_reply.started":"2024-02-24T13:48:12.432848Z","shell.execute_reply":"2024-02-24T13:48:12.445987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:12.448135Z","iopub.execute_input":"2024-02-24T13:48:12.448398Z","iopub.status.idle":"2024-02-24T13:48:12.497979Z","shell.execute_reply.started":"2024-02-24T13:48:12.448348Z","shell.execute_reply":"2024-02-24T13:48:12.497375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"totalTrain = len(list(paths.list_images(trainPath)))\ntotalVal = len(list(paths.list_images(valPath)))\ntotalTest = len(list(paths.list_images(testPath)))\n\nprint(\"[INFO] training head...\")\nH = model.fit(\n    trainGen,\n    steps_per_epoch=totalTrain // BATCH_SIZE,\n    validation_data=valGen,\n    validation_steps=totalVal // BATCH_SIZE,\n    epochs=trainEpochs)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T13:48:12.499224Z","iopub.execute_input":"2024-02-24T13:48:12.499547Z","iopub.status.idle":"2024-02-24T14:00:43.913569Z","shell.execute_reply.started":"2024-02-24T13:48:12.499485Z","shell.execute_reply":"2024-02-24T14:00:43.912501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network head...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\n\n\nplot_training(H, trainEpochs, WARMUP_PLOT_PATH)\nplt.show()  ","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:00:43.915162Z","iopub.execute_input":"2024-02-24T14:00:43.915660Z","iopub.status.idle":"2024-02-24T14:00:56.026419Z","shell.execute_reply.started":"2024-02-24T14:00:43.915604Z","shell.execute_reply":"2024-02-24T14:00:56.025196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:00:56.028698Z","iopub.execute_input":"2024-02-24T14:00:56.029109Z","iopub.status.idle":"2024-02-24T14:00:57.694119Z","shell.execute_reply.started":"2024-02-24T14:00:56.029041Z","shell.execute_reply":"2024-02-24T14:00:57.693054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns ","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:00:57.696302Z","iopub.execute_input":"2024-02-24T14:00:57.696648Z","iopub.status.idle":"2024-02-24T14:00:57.701984Z","shell.execute_reply.started":"2024-02-24T14:00:57.696582Z","shell.execute_reply":"2024-02-24T14:00:57.700951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming testGen and model are defined\ntestGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\ny_true = testGen.classes\n\n# Compute ROC curve for the positive class\nfpr, tpr, _ = roc_curve(y_true, y_pred_probs[:, 1])\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=1, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:00:57.703145Z","iopub.execute_input":"2024-02-24T14:00:57.703375Z","iopub.status.idle":"2024-02-24T14:01:09.695502Z","shell.execute_reply.started":"2024-02-24T14:00:57.703336Z","shell.execute_reply":"2024-02-24T14:01:09.694284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:01:09.697554Z","iopub.execute_input":"2024-02-24T14:01:09.698246Z","iopub.status.idle":"2024-02-24T14:01:21.382753Z","shell.execute_reply.started":"2024-02-24T14:01:09.698174Z","shell.execute_reply":"2024-02-24T14:01:21.381940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:01:21.384095Z","iopub.execute_input":"2024-02-24T14:01:21.384355Z","iopub.status.idle":"2024-02-24T14:01:21.678569Z","shell.execute_reply.started":"2024-02-24T14:01:21.384313Z","shell.execute_reply":"2024-02-24T14:01:21.677487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:01:21.680367Z","iopub.execute_input":"2024-02-24T14:01:21.680970Z","iopub.status.idle":"2024-02-24T14:01:22.036644Z","shell.execute_reply.started":"2024-02-24T14:01:21.680902Z","shell.execute_reply":"2024-02-24T14:01:22.035462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainGen.reset()\nvalGen.reset()","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:01:22.038814Z","iopub.execute_input":"2024-02-24T14:01:22.039689Z","iopub.status.idle":"2024-02-24T14:01:22.046208Z","shell.execute_reply.started":"2024-02-24T14:01:22.039615Z","shell.execute_reply":"2024-02-24T14:01:22.044609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers[15:]:\n\tlayer.trainable = True\n\n# loop over the layers in the model and show which ones are trainable\n# or not\nfor layer in baseModel.layers:\n\tprint(\"{}: {}\".format(layer, layer.trainable))\n\n# for the changes to the model to take affect we need to recompile\n# the model, this time using SGD with a *very* small learning rate\nprint(\"[INFO] re-compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:01:22.048545Z","iopub.execute_input":"2024-02-24T14:01:22.049262Z","iopub.status.idle":"2024-02-24T14:01:22.118388Z","shell.execute_reply.started":"2024-02-24T14:01:22.048897Z","shell.execute_reply":"2024-02-24T14:01:22.117469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"H = model.fit_generator(\n\ttrainGen,\n\tsteps_per_epoch=totalTrain // BATCH_SIZE,\n\tvalidation_data=valGen,\n\tvalidation_steps=totalVal // BATCH_SIZE,\n\tepochs= epochsFineTune)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:01:22.122454Z","iopub.execute_input":"2024-02-24T14:01:22.122717Z","iopub.status.idle":"2024-02-24T14:14:11.531190Z","shell.execute_reply.started":"2024-02-24T14:01:22.122666Z","shell.execute_reply":"2024-02-24T14:14:11.530107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\nplot_training(H, epochsFineTune, UNFROZEN_PLOT_PATH)\n\n# serialize the model to disk\nprint(\"[INFO] serializing network...\")\nmodel.save(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:14:11.536271Z","iopub.execute_input":"2024-02-24T14:14:11.537524Z","iopub.status.idle":"2024-02-24T14:14:23.841023Z","shell.execute_reply.started":"2024-02-24T14:14:11.537458Z","shell.execute_reply":"2024-02-24T14:14:23.838362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:14:23.843050Z","iopub.execute_input":"2024-02-24T14:14:23.843573Z","iopub.status.idle":"2024-02-24T14:14:24.193987Z","shell.execute_reply.started":"2024-02-24T14:14:23.843466Z","shell.execute_reply":"2024-02-24T14:14:24.192905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:14:24.196257Z","iopub.execute_input":"2024-02-24T14:14:24.196684Z","iopub.status.idle":"2024-02-24T14:14:24.484600Z","shell.execute_reply.started":"2024-02-24T14:14:24.196612Z","shell.execute_reply":"2024-02-24T14:14:24.482979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:14:24.486668Z","iopub.execute_input":"2024-02-24T14:14:24.487272Z","iopub.status.idle":"2024-02-24T14:14:24.757274Z","shell.execute_reply.started":"2024-02-24T14:14:24.487030Z","shell.execute_reply":"2024-02-24T14:14:24.756301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wave\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-02-24T14:14:24.758958Z","iopub.execute_input":"2024-02-24T14:14:24.759274Z","iopub.status.idle":"2024-02-24T14:14:24.776888Z","shell.execute_reply.started":"2024-02-24T14:14:24.759223Z","shell.execute_reply":"2024-02-24T14:14:24.776107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}