{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":18147,"sourceType":"datasetVersion","datasetId":13405},{"sourceId":842050,"sourceType":"datasetVersion","datasetId":444558},{"sourceId":893807,"sourceType":"datasetVersion","datasetId":451078},{"sourceId":6358196,"sourceType":"datasetVersion","datasetId":3579787},{"sourceId":8727823,"sourceType":"datasetVersion","datasetId":5238165}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-21T05:51:39.025961Z","iopub.execute_input":"2024-06-21T05:51:39.026387Z","iopub.status.idle":"2024-06-21T05:51:39.031875Z","shell.execute_reply.started":"2024-06-21T05:51:39.026316Z","shell.execute_reply":"2024-06-21T05:51:39.031092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install numpy --upgrade\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:51:39.033974Z","iopub.execute_input":"2024-06-21T05:51:39.034302Z","iopub.status.idle":"2024-06-21T05:51:46.628631Z","shell.execute_reply.started":"2024-06-21T05:51:39.034247Z","shell.execute_reply":"2024-06-21T05:51:46.627599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!pip install torch==1.10.2\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:51:46.630472Z","iopub.execute_input":"2024-06-21T05:51:46.630778Z","iopub.status.idle":"2024-06-21T05:51:52.707443Z","shell.execute_reply.started":"2024-06-21T05:51:46.630723Z","shell.execute_reply":"2024-06-21T05:51:52.70649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install Pillow==8.4.0\n!pip install torch==1.10.2\n!pip install facenet-pytorch==2.5.2\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:51:52.709818Z","iopub.execute_input":"2024-06-21T05:51:52.710204Z","iopub.status.idle":"2024-06-21T05:52:11.163987Z","shell.execute_reply.started":"2024-06-21T05:51:52.710139Z","shell.execute_reply":"2024-06-21T05:52:11.162991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!pip install facenet-pytorch","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:11.167852Z","iopub.execute_input":"2024-06-21T05:52:11.168153Z","iopub.status.idle":"2024-06-21T05:52:17.37781Z","shell.execute_reply.started":"2024-06-21T05:52:11.168098Z","shell.execute_reply":"2024-06-21T05:52:17.376866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\nimport numpy as np\nimport pandas as pd\nimport matplotlib\nimport seaborn as sns\nimport torch\nimport matplotlib.pyplot as plt\n# from tqdm import tqdm_notebook\n%matplotlib inline \n# from google.colab.patches import cv2_imshow\nfrom IPython.display import HTML #imports to play videos\nfrom base64 import b64encode \nimport cv2 as cv\nfrom skimage.measure import compare_ssim\nimport glob\nimport time\nfrom PIL import Image\nfrom facenet_pytorch import MTCNN, InceptionResnetV1, extract_face\nfrom tqdm import tqdm\n\nimport math\nimport pickle\nfrom functools import partial\nfrom collections import defaultdict\n\nfrom PIL import Image\nfrom glob import glob\n\nimport cv2\nimport skimage.measure\nimport albumentations as A\nfrom tqdm.notebook import tqdm \nfrom albumentations.pytorch import ToTensor \n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\nfrom torchvision.models.video import mc3_18, r2plus1d_18\n\nfrom facenet_pytorch import MTCNN","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.38116Z","iopub.execute_input":"2024-06-21T05:52:17.381467Z","iopub.status.idle":"2024-06-21T05:52:17.400024Z","shell.execute_reply.started":"2024-06-21T05:52:17.381414Z","shell.execute_reply":"2024-06-21T05:52:17.399299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = \"../input/deepfake-detection-challenge\" \nTRAIN_SAMPLE_FOLDER = \"train_sample_videos\"\nTEST_FOLDER = \"test_videos\"","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.401592Z","iopub.execute_input":"2024-06-21T05:52:17.401892Z","iopub.status.idle":"2024-06-21T05:52:17.41438Z","shell.execute_reply.started":"2024-06-21T05:52:17.401841Z","shell.execute_reply":"2024-06-21T05:52:17.413697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '../input/haarcascades'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")    ","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.415569Z","iopub.execute_input":"2024-06-21T05:52:17.415783Z","iopub.status.idle":"2024-06-21T05:52:17.426126Z","shell.execute_reply.started":"2024-06-21T05:52:17.415747Z","shell.execute_reply":"2024-06-21T05:52:17.425244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\") ","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.427591Z","iopub.execute_input":"2024-06-21T05:52:17.427884Z","iopub.status.idle":"2024-06-21T05:52:17.435712Z","shell.execute_reply.started":"2024-06-21T05:52:17.42783Z","shell.execute_reply":"2024-06-21T05:52:17.435071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.436916Z","iopub.execute_input":"2024-06-21T05:52:17.437212Z","iopub.status.idle":"2024-06-21T05:52:17.445602Z","shell.execute_reply.started":"2024-06-21T05:52:17.43715Z","shell.execute_reply":"2024-06-21T05:52:17.444803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")\n#reading the json file\ndef get_meta_from_json(path):\n    df = pd.read_json(os.path.join(DATA_FOLDER, path, json_file))\n    df = df.T\n    return df\n\nmeta_train_df = get_meta_from_json(TRAIN_SAMPLE_FOLDER)\nmeta_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.446869Z","iopub.execute_input":"2024-06-21T05:52:17.447176Z","iopub.status.idle":"2024-06-21T05:52:17.633776Z","shell.execute_reply.started":"2024-06-21T05:52:17.447126Z","shell.execute_reply":"2024-06-21T05:52:17.632921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.635102Z","iopub.execute_input":"2024-06-21T05:52:17.635365Z","iopub.status.idle":"2024-06-21T05:52:17.642836Z","shell.execute_reply.started":"2024-06-21T05:52:17.635316Z","shell.execute_reply":"2024-06-21T05:52:17.641871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.64424Z","iopub.execute_input":"2024-06-21T05:52:17.644505Z","iopub.status.idle":"2024-06-21T05:52:17.664091Z","shell.execute_reply.started":"2024-06-21T05:52:17.644452Z","shell.execute_reply":"2024-06-21T05:52:17.663312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df.loc[meta_train_df.label == 'REAL'])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.665575Z","iopub.execute_input":"2024-06-21T05:52:17.665894Z","iopub.status.idle":"2024-06-21T05:52:17.682377Z","shell.execute_reply.started":"2024-06-21T05:52:17.66584Z","shell.execute_reply":"2024-06-21T05:52:17.681543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Totals']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique() #collect all unique instances\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.683738Z","iopub.execute_input":"2024-06-21T05:52:17.684067Z","iopub.status.idle":"2024-06-21T05:52:17.692294Z","shell.execute_reply.started":"2024-06-21T05:52:17.684013Z","shell.execute_reply":"2024-06-21T05:52:17.691507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.693378Z","iopub.execute_input":"2024-06-21T05:52:17.693625Z","iopub.status.idle":"2024-06-21T05:52:17.710361Z","shell.execute_reply.started":"2024-06-21T05:52:17.693579Z","shell.execute_reply":"2024-06-21T05:52:17.70954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_frequent_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    items = []\n    vals = []\n    for col in data.columns:\n        itm = data[col].value_counts().index[0]\n        val = data[col].value_counts().values[0]\n        items.append(itm)\n        vals.append(val)\n    tt['Most frequent item'] = items\n    tt['Frequence'] = vals\n    tt['Percent from total'] = np.round(vals / total * 100, 3)\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.711497Z","iopub.execute_input":"2024-06-21T05:52:17.711769Z","iopub.status.idle":"2024-06-21T05:52:17.719394Z","shell.execute_reply.started":"2024-06-21T05:52:17.711699Z","shell.execute_reply":"2024-06-21T05:52:17.718473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.720881Z","iopub.execute_input":"2024-06-21T05:52:17.721235Z","iopub.status.idle":"2024-06-21T05:52:17.745866Z","shell.execute_reply.started":"2024-06-21T05:52:17.72118Z","shell.execute_reply":"2024-06-21T05:52:17.745219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df.loc[meta_train_df.label == 'FAKE'])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.747309Z","iopub.execute_input":"2024-06-21T05:52:17.747672Z","iopub.status.idle":"2024-06-21T05:52:17.774443Z","shell.execute_reply.started":"2024-06-21T05:52:17.747627Z","shell.execute_reply":"2024-06-21T05:52:17.77374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n  '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n  '''  \n  f, ax = plt.subplots(1,1, figsize=(4*size,4))\n  total = float(len(df))\n  g =  sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n  g.set_title(\"Number and percentage of {}\".format(title)) \n  if(size > 2):\n    plt.xticks(rotation=90, size=8)\n  for p in ax.patches:\n     height = p.get_height()\n     ax.text(p.get_x()+ p.get_width()/2.,height + 3,'{:1.2f}%'.format(100*height/total),ha=\"center\")\n\n  plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.77589Z","iopub.execute_input":"2024-06-21T05:52:17.776214Z","iopub.status.idle":"2024-06-21T05:52:17.785778Z","shell.execute_reply.started":"2024-06-21T05:52:17.776159Z","shell.execute_reply":"2024-06-21T05:52:17.785112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('split','split(train)',meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:17.787479Z","iopub.execute_input":"2024-06-21T05:52:17.787784Z","iopub.status.idle":"2024-06-21T05:52:18.011568Z","shell.execute_reply.started":"2024-06-21T05:52:17.787732Z","shell.execute_reply":"2024-06-21T05:52:18.010539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('label','label(train)',meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:18.013621Z","iopub.execute_input":"2024-06-21T05:52:18.01439Z","iopub.status.idle":"2024-06-21T05:52:18.255707Z","shell.execute_reply.started":"2024-06-21T05:52:18.01432Z","shell.execute_reply":"2024-06-21T05:52:18.25467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = np.array(list(meta_train_df.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:18.257743Z","iopub.execute_input":"2024-06-21T05:52:18.258493Z","iopub.status.idle":"2024-06-21T05:52:18.272995Z","shell.execute_reply.started":"2024-06-21T05:52:18.258423Z","shell.execute_reply":"2024-06-21T05:52:18.271803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='FAKE'].sample(3).index)\nfake_train_sample_video\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:18.27518Z","iopub.execute_input":"2024-06-21T05:52:18.275978Z","iopub.status.idle":"2024-06-21T05:52:18.287724Z","shell.execute_reply.started":"2024-06-21T05:52:18.275868Z","shell.execute_reply":"2024-06-21T05:52:18.286679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_img = cv.VideoCapture(video_path)\n    ret, frame = capture_img.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:18.290179Z","iopub.execute_input":"2024-06-21T05:52:18.290938Z","iopub.status.idle":"2024-06-21T05:52:18.301509Z","shell.execute_reply.started":"2024-06-21T05:52:18.290867Z","shell.execute_reply":"2024-06-21T05:52:18.299715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video_file))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:18.30375Z","iopub.execute_input":"2024-06-21T05:52:18.304693Z","iopub.status.idle":"2024-06-21T05:52:20.083682Z","shell.execute_reply.started":"2024-06-21T05:52:18.304613Z","shell.execute_reply":"2024-06-21T05:52:20.082888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='REAL'].sample(3).index) #viewing the real videos\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:20.085012Z","iopub.execute_input":"2024-06-21T05:52:20.085236Z","iopub.status.idle":"2024-06-21T05:52:20.092393Z","shell.execute_reply.started":"2024-06-21T05:52:20.085198Z","shell.execute_reply":"2024-06-21T05:52:20.091611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video in real_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:20.093996Z","iopub.execute_input":"2024-06-21T05:52:20.094281Z","iopub.status.idle":"2024-06-21T05:52:21.910677Z","shell.execute_reply.started":"2024-06-21T05:52:20.094233Z","shell.execute_reply":"2024-06-21T05:52:21.909537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    #we only show images extracted from first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n      video_path = os.path.join(DATA_FOLDER, video_folder, video_file)\n      capture_img = cv.VideoCapture(video_path)\n      ret, frame = capture_img.read()\n      frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n      ax[i//3, i%3].imshow(frame)\n      ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n      ax[i//3, i%3].axis('on')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:21.912806Z","iopub.execute_input":"2024-06-21T05:52:21.91354Z","iopub.status.idle":"2024-06-21T05:52:21.923276Z","shell.execute_reply.started":"2024-06-21T05:52:21.913468Z","shell.execute_reply":"2024-06-21T05:52:21.922525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='meawmsgiti.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:21.924711Z","iopub.execute_input":"2024-06-21T05:52:21.924974Z","iopub.status.idle":"2024-06-21T05:52:24.660108Z","shell.execute_reply.started":"2024-06-21T05:52:21.924928Z","shell.execute_reply":"2024-06-21T05:52:24.659282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvf /kaggle/input/ffmpeg-static-build/ffmpeg-git-amd64-static.tar.xz","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:24.661514Z","iopub.execute_input":"2024-06-21T05:52:24.661813Z","iopub.status.idle":"2024-06-21T05:52:29.079977Z","shell.execute_reply.started":"2024-06-21T05:52:24.661738Z","shell.execute_reply":"2024-06-21T05:52:29.079072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob, shutil\nimport timeit, os, gc\nimport subprocess as sp\nfrom tqdm import tqdm\nfrom collections import defaultdict\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nimport json\nfrom IPython.display import HTML\nfrom base64 import b64encode\nimport cv2\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:29.090984Z","iopub.execute_input":"2024-06-21T05:52:29.091239Z","iopub.status.idle":"2024-06-21T05:52:29.10183Z","shell.execute_reply.started":"2024-06-21T05:52:29.091202Z","shell.execute_reply":"2024-06-21T05:52:29.100982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOME = \"./\"\nFFMPEG = \"/kaggle/working/ffmpeg-git-20191209-amd64-static\"\nFFMPEG_PATH = FFMPEG\nDATA_FOLDER = \"/kaggle/input/deepfake-detection-challenge\"\nTMP_FOLDER = HOME\nDATA_FOLDER_TRAIN = DATA_FOLDER\nVIDEOS_FOLDER_TRAIN = DATA_FOLDER_TRAIN + \"/train_sample_videos\"\nIMAGES_FOLDER_TRAIN = TMP_FOLDER + \"/images\"\nAUDIOS_FOLDER_TRAIN = TMP_FOLDER + \"/audios\"\nEXTRACT_META = True # False\nEXTRACT_CONTENT = True # False\nEXTRACT_FACES = True # False\nFRAME_RATE = 0.5 # Frame per\nprint(FFMPEG)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:29.103683Z","iopub.execute_input":"2024-06-21T05:52:29.103941Z","iopub.status.idle":"2024-06-21T05:52:29.112528Z","shell.execute_reply.started":"2024-06-21T05:52:29.103866Z","shell.execute_reply":"2024-06-21T05:52:29.111619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_command(*popenargs, **kwargs):\n    closeNULL = 0\n    try:\n        from subprocess import DEVNULL\n        closeNULL = 0\n    except ImportError:\n        import os\n        DEVNULL = open(os.devnull, 'wb')\n        closeNULL = 1\n\n    process = sp.Popen(stdout=sp.PIPE, stderr=DEVNULL, *popenargs, **kwargs)\n    output, unused_err = process.communicate()\n    retcode = process.poll()\n\n    if closeNULL:\n        DEVNULL.close()\n\n    if retcode:\n        cmd = kwargs.get(\"args\")\n        if cmd is None:\n            cmd = popenargs[0]\n        error = sp.CalledProcessError(retcode, cmd)\n        error.output = output\n        raise error\n    return output\n\ndef ffprobe(filename, options = [\"-show_error\", \"-show_format\", \"-show_streams\", \"-show_programs\", \"-show_chapters\", \"-show_private_data\"]):\n    ret = {}\n    command = [FFMPEG_PATH + \"/ffprobe\", \"-v\", \"error\", *options, \"-print_format\", \"json\", filename]\n    ret = run_command(command)\n    if ret:\n        ret = json.loads(ret)\n    return ret\n\n# ffmpeg -i input.mov -r 0.25 output_%04d.png\ndef ffextract_frames(filename, output_folder, rate = 0.25):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-r\", str(rate), \"-y\", output_folder + \"/output_%04d.png\"]\n    ret = run_command(command)\n    return ret\n\n# ffmpeg -i input-video.mp4 output-audio.mp3\ndef ffextract_audio(filename, output_path):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-vn\", \"-ac\", \"1\", \"-acodec\", \"copy\", \"-y\", output_path]\n    ret = run_command(command)\n    return ret","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:29.113828Z","iopub.execute_input":"2024-06-21T05:52:29.114086Z","iopub.status.idle":"2024-06-21T05:52:29.128866Z","shell.execute_reply.started":"2024-06-21T05:52:29.114038Z","shell.execute_reply":"2024-06-21T05:52:29.128095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTRACT_META == True:\n    results = []\n    subfolder = VIDEOS_FOLDER_TRAIN\n    filepaths = glob.glob(subfolder + \"/*.mp4\")\n    for filepath in tqdm(filepaths):\n        js = ffprobe(filepath)\n#        print(js)\n        if js:\n            results.append(\n                (js.get(\"format\", {}).get(\"filename\")[len(subfolder) + 1:],\n                js.get(\"format\", {}).get(\"format_long_name\"),\n                # Video \n                js.get(\"streams\", [{}, {}])[0].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"height\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"width\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"start_time\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"avg_frame_rate\"),\n                 # Audio\n                js.get(\"streams\", [{}, {}])[1].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"channels\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"sample_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"start_time\")),\n            )\n\n    meta_pd = pd.DataFrame(results, columns=[\"filename\", \"format\", \"video_codec_name\", \"video_height\", \"video_width\",\n                                            \"video_nb_frames\", \"video_bit_rate\", \"video_duration\", \"video_start_time\",\"video_fps\",\n                                            \"audio_codec_name\", \"audio_channels\", \"audio_sample_rate\", \"audio_nb_frames\",\n                                            \"audio_bit_rate\", \"audio_duration\", \"audio_start_time\"])\n    meta_pd[\"video_fps\"] = meta_pd[\"video_fps\"].apply(lambda x: float(x.split(\"/\")[0])/float(x.split(\"/\")[1]) if len(x.split(\"/\")) == 2 else None)\n    meta_pd[\"video_duration\"] = meta_pd[\"video_duration\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"video_start_time\"] = meta_pd[\"video_start_time\"].astype(np.float32)\n    meta_pd[\"video_nb_frames\"] = meta_pd[\"video_nb_frames\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_sample_rate\"] = meta_pd[\"audio_sample_rate\"].astype(np.float32)\n    meta_pd[\"audio_nb_frames\"] = meta_pd[\"audio_nb_frames\"].astype(np.float32)\n    meta_pd[\"audio_bit_rate\"] = meta_pd[\"audio_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_duration\"] = meta_pd[\"audio_duration\"].astype(np.float32)\n    meta_pd[\"audio_start_time\"] = meta_pd[\"audio_start_time\"].astype(np.float32)\n    meta_pd.to_pickle(HOME + \"videos_meta.pkl\")\nelse:\n    meta_pd = pd.read_pickle(HOME + \"videos_meta.pkl\")\nmeta_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:52:29.130445Z","iopub.execute_input":"2024-06-21T05:52:29.130761Z","iopub.status.idle":"2024-06-21T05:53:07.602432Z","shell.execute_reply.started":"2024-06-21T05:52:29.130707Z","shell.execute_reply":"2024-06-21T05:53:07.601401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,6, figsize=(22, 3))\nd = sns.distplot(meta_pd[\"video_fps\"], ax=ax[0])\nd = sns.distplot(meta_pd[\"video_duration\"], ax=ax[1])\nd = sns.distplot(meta_pd[\"video_width\"], ax=ax[2])\nd = sns.distplot(meta_pd[\"video_height\"], ax=ax[3])\nd = sns.distplot(meta_pd[\"video_nb_frames\"], ax=ax[4])\nd = sns.distplot(meta_pd[\"video_bit_rate\"], ax=ax[5])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:53:07.603981Z","iopub.execute_input":"2024-06-21T05:53:07.604231Z","iopub.status.idle":"2024-06-21T05:53:09.436894Z","shell.execute_reply.started":"2024-06-21T05:53:07.604189Z","shell.execute_reply":"2024-06-21T05:53:09.436073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:53:09.438366Z","iopub.execute_input":"2024-06-21T05:53:09.438905Z","iopub.status.idle":"2024-06-21T05:53:09.618677Z","shell.execute_reply.started":"2024-06-21T05:53:09.438851Z","shell.execute_reply":"2024-06-21T05:53:09.617974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:53:09.61972Z","iopub.execute_input":"2024-06-21T05:53:09.619976Z","iopub.status.idle":"2024-06-21T05:53:09.792047Z","shell.execute_reply.started":"2024-06-21T05:53:09.619927Z","shell.execute_reply":"2024-06-21T05:53:09.791374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.merge(train_pd, meta_pd[[\"filename\", \"video_height\", \"video_width\", \"video_nb_frames\", \"video_bit_rate\", \"audio_nb_frames\"]], on=\"filename\", how=\"left\")\ntrain_pd[\"count\"] = train_pd.groupby([\"original\"])[\"original\"].transform('count')\n# train_pd.to_pickle(HOME + \"train_meta.pkl\")\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:53:09.793096Z","iopub.execute_input":"2024-06-21T05:53:09.793314Z","iopub.status.idle":"2024-06-21T05:53:09.821833Z","shell.execute_reply.started":"2024-06-21T05:53:09.793277Z","shell.execute_reply":"2024-06-21T05:53:09.821148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUDIO_FORMAT = \"aac\" # \"wav\"\nvideos_folder = VIDEOS_FOLDER_TRAIN\nimages_folder_path = IMAGES_FOLDER_TRAIN\naudios_folder_path = AUDIOS_FOLDER_TRAIN\nif EXTRACT_CONTENT == True:\n    # 1h20min for chunk#0 (11GB)\n    # Extract some images + audio track\n    for idx, row in tqdm(train_pd.iterrows(), total=meta_pd.shape[0]):\n        try:\n            video_path = videos_folder + \"/\" + row[\"filename\"]\n            images_path = images_folder_path + \"/\" + row[\"filename\"][:-4]\n            audio_path = audios_folder_path + \"/\" + row[\"filename\"][:-4]\n            # Extract images\n            if not os.path.exists(images_path): os.makedirs(images_path)\n            ret = ffextract_frames(video_path, images_path, rate = FRAME_RATE)\n            # Extract audio\n            if not os.path.exists(audio_path): os.makedirs(audio_path)\n            # ret = ffextract_audio(video_path, audio_path + \"/audio.\" + AUDIO_FORMAT)\n        except:\n            print(\"Cannot extract frames/audio for:\" + row[\"filename\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T05:53:09.82333Z","iopub.execute_input":"2024-06-21T05:53:09.823629Z","iopub.status.idle":"2024-06-21T06:05:31.627284Z","shell.execute_reply.started":"2024-06-21T05:53:09.823576Z","shell.execute_reply":"2024-06-21T06:05:31.626435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd.tail()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:31.629266Z","iopub.execute_input":"2024-06-21T06:05:31.629577Z","iopub.status.idle":"2024-06-21T06:05:31.649297Z","shell.execute_reply.started":"2024-06-21T06:05:31.629525Z","shell.execute_reply":"2024-06-21T06:05:31.648551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 12\nfake = train_pd[\"filename\"][idx]\nreal = train_pd[\"original\"][idx]\nvid_width = train_pd[\"video_width\"][idx]\nvid_real = open(VIDEOS_FOLDER_TRAIN + \"/\" + real, 'rb').read()\ndata_url_real = \"data:video/mp4;base64,\" + b64encode(vid_real).decode()\nvid_fake = open(VIDEOS_FOLDER_TRAIN + \"/\" + fake, 'rb').read()\ndata_url_fake = \"data:video/mp4;base64,\" + b64encode(vid_fake).decode()\nHTML(\"\"\"\n<div style='width: 100%%; display: table;'>\n    <div style='display: table-row'>\n        <div style='width: %dpx; display: table-cell;'><b>Real</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n        <div style='display: table-cell;'><b>Fake</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n    </div>\n</div>\n\"\"\" % ( int(vid_width/3.2) + 10, \n       real, int(vid_width/3.2), data_url_real, \n       fake, int(vid_width/3.2), data_url_fake))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:31.650679Z","iopub.execute_input":"2024-06-21T06:05:31.65099Z","iopub.status.idle":"2024-06-21T06:05:31.781277Z","shell.execute_reply.started":"2024-06-21T06:05:31.650939Z","shell.execute_reply":"2024-06-21T06:05:31.780027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + \"haarcascade_frontalface_default.xml\")\n\ndef detect_face_cv2(img):\n    # Move to grayscale\n    gray_img = cv2.cvtColor(img.copy(), cv2.COLOR_RGB2GRAY)\n    face_locations = []\n    face_rects = face_cascade.detectMultiScale(gray_img, scaleFactor=1.3, minNeighbors=5)     \n    for (x,y,w,h) in face_rects: \n        face_location = (x,y,w,h)\n        face_locations.append((face_location, 1.0))\n    return face_locations","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:31.783455Z","iopub.execute_input":"2024-06-21T06:05:31.783919Z","iopub.status.idle":"2024-06-21T06:05:31.835992Z","shell.execute_reply.started":"2024-06-21T06:05:31.783831Z","shell.execute_reply":"2024-06-21T06:05:31.835244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mtcnn","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:31.837283Z","iopub.execute_input":"2024-06-21T06:05:31.837558Z","iopub.status.idle":"2024-06-21T06:05:38.021087Z","shell.execute_reply.started":"2024-06-21T06:05:31.837509Z","shell.execute_reply":"2024-06-21T06:05:38.020151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mtcnn import MTCNN\ndetector = MTCNN()\n\ndef detect_face_mtcnn(img):\n    face_locations = []\n    items = detector.detect_faces(img)\n    for face in items:\n        face_location = tuple(face.get('box'))\n        face_confidence = float(face.get('confidence'))\n        face_locations.append((face_location, face_confidence))\n    return face_locations","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:38.022879Z","iopub.execute_input":"2024-06-21T06:05:38.023157Z","iopub.status.idle":"2024-06-21T06:05:38.349452Z","shell.execute_reply.started":"2024-06-21T06:05:38.023112Z","shell.execute_reply":"2024-06-21T06:05:38.348445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_faces(files, source, detector=detect_face_cv2):\n    results = []\n    # for idx, file in tqdm(enumerate(files), total=len(files)):\n    for idx, file in enumerate(files):\n        try:\n            img = cv2.cvtColor(cv2.imread(file, cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)\n            face_locations = detector(img)\n            results.append((source, file[file.find(\"output_\"):], face_locations, len(face_locations)))\n        except:\n            print(\"Cannot extract faces for image: %s\" % file)\n    return results","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:38.350906Z","iopub.execute_input":"2024-06-21T06:05:38.351201Z","iopub.status.idle":"2024-06-21T06:05:38.358464Z","shell.execute_reply.started":"2024-06-21T06:05:38.351147Z","shell.execute_reply":"2024-06-21T06:05:38.357613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = fake\ndump_folder = images_folder_path + \"/\" + file[:-4]\nfiles = glob.glob(dump_folder + \"/*\")\nDETECTORS = {\n    \"cv2\": detect_face_cv2,\n    \"mtcnn\": detect_face_mtcnn\n}\nfaces_pd = None\nfor key, value in DETECTORS.items():\n    tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n    if faces_pd is None:\n        faces_pd = tmp_pd\n    else:\n        faces_pd = pd.merge(faces_pd, tmp_pd, on=[\"filename\", \"image\"], how=\"left\")\nfaces_pd.head(12)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:38.359771Z","iopub.execute_input":"2024-06-21T06:05:38.360268Z","iopub.status.idle":"2024-06-21T06:05:44.667445Z","shell.execute_reply.started":"2024-06-21T06:05:38.359983Z","shell.execute_reply":"2024-06-21T06:05:44.666682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_faces_boxes(df, max_cols = 2, max_rows = 6, fsize=(24, 5), max_items=12):    \n    idx = 0    \n    for item_idx, item in df.iterrows():\n        img = cv2.cvtColor(cv2.imread(IMAGES_FOLDER_TRAIN + \"/\" + item[\"filename\"][:-4] +\"/\" + item[\"image\"], cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)    \n        face_img = img #.copy()\n        # grid subplots\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0: fig = plt.figure(figsize=fsize)\n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        # display image with boxes\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols, 0):\n            face_locations = item[c]\n            face_confidence = item[c]            \n            if len(face_locations) > 0:\n                for face_location in face_locations:        \n                    ((x,y,w,h), confidence) = face_location\n                    # face_img = face_img[y:y+h, x:x+w]\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255,i*255,0), 8)\n                    cv2.putText(face_img, '%.1f' % (confidence*100.0), (x+w, y+h), cv2.FONT_HERSHEY_SIMPLEX, 2.0, (255,i*255,0), 9, cv2.LINE_AA)\n                ax.imshow(face_img)\n            else:\n                ax.imshow(img)\n            ax.set_title(\"%s %s / %s - Faces: %d %s %s\" % (item[\"label\"] if \"label\" in df.columns else \"\", \n                                                           item[\"filename\"], item[\"image\"],\n                                                           item[\"faces_mtcnn\"] if \"faces_mtcnn\" in df.columns else len(face_locations),\n                                                           item[\"faces_mtcnn_median\"] if \"faces_mtcnn_median\" in df.columns else \"\",\n                                                           item[\"faces\"] if \"faces\" in df.columns else \"\"))\n        if (col == max_cols -1): plt.show()\n        idx = idx + 1\n        if (max_items > 0 and idx >=max_items): break","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:44.668992Z","iopub.execute_input":"2024-06-21T06:05:44.669281Z","iopub.status.idle":"2024-06-21T06:05:44.68667Z","shell.execute_reply.started":"2024-06-21T06:05:44.66923Z","shell.execute_reply":"2024-06-21T06:05:44.685977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(faces_pd)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:44.688092Z","iopub.execute_input":"2024-06-21T06:05:44.688356Z","iopub.status.idle":"2024-06-21T06:05:47.225014Z","shell.execute_reply.started":"2024-06-21T06:05:44.688317Z","shell.execute_reply":"2024-06-21T06:05:47.224213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update the conda environment and install cmake\n!conda install -c conda-forge cmake\n!conda install -c conda-forge dlib","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:05:47.226369Z","iopub.execute_input":"2024-06-21T06:05:47.226655Z","iopub.status.idle":"2024-06-21T06:06:07.118549Z","shell.execute_reply.started":"2024-06-21T06:05:47.2266Z","shell.execute_reply":"2024-06-21T06:06:07.117718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install mutils\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:06:07.120295Z","iopub.execute_input":"2024-06-21T06:06:07.12053Z","iopub.status.idle":"2024-06-21T06:06:13.32312Z","shell.execute_reply.started":"2024-06-21T06:06:07.120489Z","shell.execute_reply":"2024-06-21T06:06:13.322027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update\n!apt-get install -y build-essential cmake\n!pip install wheel\n\n!pip install dlib==19.24.0","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:06:13.324784Z","iopub.execute_input":"2024-06-21T06:06:13.325099Z","iopub.status.idle":"2024-06-21T06:06:29.646185Z","shell.execute_reply.started":"2024-06-21T06:06:13.325039Z","shell.execute_reply":"2024-06-21T06:06:29.64515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the facial landmarks for the different facial regions\nFACIAL_LANDMARKS_INDEXES = {\n    \"Jaw\": (0, 17),\n    \"Right Eyebrow\": (17, 22),\n    \"Left Eyebrow\": (22, 27),\n    \"Nose\": (27, 36),\n    \"Right Eye\": (36, 42),\n    \"Left Eye\": (42, 48),\n    \"Mouth\": (48, 68)\n}\n\nimport cv2\nimport matplotlib.pyplot as plt\nimport dlib\nimport numpy as np\n\ndef shape_to_numpy_array(shape, dtype=\"int\"):\n    # Convert dlib's shape object to a numpy array\n    coordinates = np.zeros((68, 2), dtype=dtype)\n    for i in range(0, 68):\n        coordinates[i] = (shape.part(i).x, shape.part(i).y)\n    return coordinates\n\ndef visualize_facial_landmarks(image, shape, colors=None, alpha=0.75):\n    # Create overlay and output images\n    overlay = image.copy()\n    output = image.copy()\n\n    if colors is None:\n        colors = [(19, 199, 109), (79, 76, 240), (230, 159, 23),\n                  (168, 100, 168), (158, 163, 32),\n                  (163, 38, 32), (180, 42, 220)]\n\n    for (i, name) in enumerate(FACIAL_LANDMARKS_INDEXES.keys()):\n        (j, k) = FACIAL_LANDMARKS_INDEXES[name]\n        pts = shape[j:k]\n\n        if name == \"Jaw\":\n            for l in range(1, len(pts)):\n                ptA = tuple(pts[l - 1])\n                ptB = tuple(pts[l])\n                cv2.line(overlay, ptA, ptB, colors[i], 2)\n        else:\n            hull = cv2.convexHull(pts)\n            cv2.drawContours(overlay, [hull], -1, colors[i], -1)\n\n    cv2.addWeighted(overlay, alpha, output, 1 - alpha, 0, output)\n    return output\n\n# def plot_faces_boxes(df, shape_predictor_path, max_cols=2, max_rows=6, fsize=(24, 5), max_items=12):\n#     # Initialize dlib's face detector and shape predictor\n#     detector = dlib.get_frontal_face_detector()\n#     predictor = dlib.shape_predictor(shape_predictor_path)\n\n#     idx = 0\n\n#     for item_idx, item in df.iterrows():\n#         # Load the image\n#         img_path = item[IMAGES_FOLDER_TRAIN]  # Ensure 'image_path' is a column in your DataFrame\n#         img = cv2.imread(img_path)\n#         if img is None:\n#             print(f\"Image not found or cannot be loaded: {img_path}\")\n#             continue\n#         img_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n#         # Prepare the image for drawing\n#         face_img = img_rgb.copy()\n\n#         # Calculate grid position\n#         row = idx // max_cols\n#         col = idx % max_cols\n\n#         if col == 0:\n#             fig = plt.figure(figsize=fsize)\n\n#         ax = fig.add_subplot(1, max_cols, col + 1)\n#         ax.axis(\"off\")\n\n#         # Extract face boxes and confidence scores\n#         face_cols = [c for c in df.columns if \"boxes\" in c]\n#         for i, c in enumerate(face_cols):\n#             face_locations = item[c]\n#             if len(face_locations) > 0:\n#                 for face_location in face_locations:\n#                     ((x, y, w, h), confidence) = face_location\n\n#                     # Draw rectangle around the face\n#                     cv2.rectangle(face_img, (x, y), (x + w, y + h), (255, i * 255, 0), 2)\n#                     cv2.putText(face_img, f'{confidence*100:.1f}%', (x, y - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, i * 255, 0), 2, cv2.LINE_AA)\n\n#                     # Detect facial landmarks\n#                     gray = cv2.cvtColor(face_img, cv2.COLOR_RGB2GRAY)\n#                     rects = detector(gray, 1)\n#                     for (j, rect) in enumerate(rects):\n#                         shape = predictor(gray, rect)\n#                         shape = shape_to_numpy_array(shape)\n#                         face_img = visualize_facial_landmarks(face_img, shape)\n\n#                 ax.imshow(face_img)\n#             else:\n#                 ax.imshow(img_rgb)\n\n#         # Set title\n#         ax.set_title(f\"{item['filename']} / {item['image']} - Faces: {item['faces_mtcnn'] if 'faces_mtcnn' in df.columns else len(face_locations)}\")\n\n#         # Show plot if end of row or last item\n#         if col == max_cols - 1 or idx == max_items - 1:\n#             plt.show()\n\n#         idx += 1\n#         if max_items > 0 and idx >= max_items:\n#             break\ndef plot_faces_boxes(df, images_folder, shape_predictor_path, max_cols=2, max_rows=6, fsize=(24, 5), max_items=12):\n    # Initialize dlib's face detector (HOG-based) and then create the facial landmark predictor\n    detector = dlib.get_frontal_face_detector()\n    predictor = dlib.shape_predictor(shape_predictor_path)\n\n    idx = 0\n    fig = None\n\n    for item_idx, item in df.iterrows():\n        # Load image\n        img_path = f\"{images_folder}/{item['filename'][:-4]}/{item['image']}\"\n        img = cv2.cvtColor(cv2.imread(img_path, cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)\n        \n        # Copy the image to draw bounding boxes\n        face_img = img.copy()\n        \n        # Calculate grid position\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0:\n            fig = plt.figure(figsize=fsize)\n        \n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        \n        # Extract face boxes and confidence scores\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols):\n            face_locations = item[c]\n            \n            if len(face_locations) > 0:\n                for face_location in face_locations:\n                    ((x, y, w, h), confidence) = face_location\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255, i*255, 0), 8)\n                    cv2.putText(face_img, f'{confidence*100:.1f}', (x+w, y+h), cv2.FONT_HERSHEY_SIMPLEX, 2.0, (255, i*255, 0), 9, cv2.LINE_AA)\n                    \n                    # Detect facial landmarks\n                    gray = cv2.cvtColor(face_img, cv2.COLOR_BGR2GRAY)\n                    rects = detector(gray, 1)\n                    for (i, rect) in enumerate(rects):\n                        shape = predictor(gray, rect)\n                        shape = shape_to_numpy_array(shape)\n                        face_img = visualize_facial_landmarks(face_img, shape)\n                    \n                ax.imshow(face_img)\n            else:\n                ax.imshow(img)\n        \n        # Set title\n        ax.set_title(f\"{item['label'] if 'label' in df.columns else ''} {item['filename']} / {item['image']} - Faces: {item['faces_mtcnn'] if 'faces_mtcnn' in df.columns else len(face_locations)} {item['faces_mtcnn_median'] if 'faces_mtcnn_median' in df.columns else ''} {item['faces'] if 'faces' in df.columns else ''}\")\n        \n        # Show plot if end of row or last item\n        if col == max_cols - 1 or idx == max_items - 1:\n            plt.show()\n        \n        idx += 1\n        if max_items > 0 and idx >= max_items:\n            break\n\n# Example usage\n# Assuming your DataFrame faces_pd includes a column image_path with full paths to the images.\nSHAPE_PREDICTOR_PATH = \"/kaggle/input/shape-predictor-68-face-landmarks-dat/shape_predictor_68_face_landmarks.dat\"  # Path to the dlib shape predictor file\n\n# Replace 'image_path' with the actual column name if different\nplot_faces_boxes(faces_pd, IMAGES_FOLDER_TRAIN, SHAPE_PREDICTOR_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:06:29.648086Z","iopub.execute_input":"2024-06-21T06:06:29.648349Z","iopub.status.idle":"2024-06-21T06:06:48.343213Z","shell.execute_reply.started":"2024-06-21T06:06:29.648303Z","shell.execute_reply":"2024-06-21T06:06:48.342404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install imutils","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:06:48.344957Z","iopub.execute_input":"2024-06-21T06:06:48.345296Z","iopub.status.idle":"2024-06-21T06:06:54.500548Z","shell.execute_reply.started":"2024-06-21T06:06:48.345248Z","shell.execute_reply":"2024-06-21T06:06:54.499495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport dlib\nimport pandas as pd\n\n# Define the facial landmarks for different facial regions\nFACIAL_LANDMARKS_INDEXES = {\n    \"Jaw\": (0, 17),\n    \"Right Eyebrow\": (17, 22),\n    \"Left Eyebrow\": (22, 27),\n    \"Nose\": (27, 36),\n    \"Right Eye\": (36, 42),\n    \"Left Eye\": (42, 48),\n    \"Mouth\": (48, 68)\n}\n\ndef plot_faces_boxes(df, images_folder, shape_predictor_path, max_cols=2, max_rows=6, fsize=(24, 5), max_items=12):\n    # Initialize dlib's face detector (HOG-based) and shape predictor\n    detector = dlib.get_frontal_face_detector()\n    predictor = dlib.shape_predictor(shape_predictor_path)\n\n    idx = 0\n    fig = None\n\n    for item_idx, item in df.iterrows():\n        # Load image\n        img_path = f\"{images_folder}/{item['filename'][:-4]}/{item['image']}\"\n        img = cv2.imread(img_path)\n        if img is None:\n            print(f\"Error: Unable to load image {img_path}\")\n            continue\n        \n        # Copy the image to draw bounding boxes\n        face_img = img.copy()\n        \n        # Calculate grid position\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0:\n            fig = plt.figure(figsize=fsize)\n        \n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        \n        # Extract face boxes and confidence scores\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols):\n            face_locations = item[c]\n            \n            if len(face_locations) > 0:\n                for face_location in face_locations:\n                    ((x, y, w, h), confidence) = face_location\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255, i*255, 0), 2)\n                    cv2.putText(face_img, f'{confidence*100:.1f}', (x, y-10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, i*255, 0), 2)\n                    \n                    # Detect facial landmarks\n                    gray = cv2.cvtColor(face_img, cv2.COLOR_BGR2GRAY)\n                    rects = detector(gray, 1)\n                    for (i, rect) in enumerate(rects):\n                        shape = predictor(gray, rect)\n                        for (name, (i, j)) in FACIAL_LANDMARKS_INDEXES.items():\n                            pts = shape_to_numpy_array(shape)[i:j]\n                            cv2.rectangle(face_img, (pts[:, 0].min(), pts[:, 1].min()), (pts[:, 0].max(), pts[:, 1].max()), (255, 0, 0), 2)\n                    \n                ax.imshow(cv2.cvtColor(face_img, cv2.COLOR_BGR2RGB))  # Convert to RGB for displaying in matplotlib\n            else:\n                ax.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))  # Convert to RGB for displaying in matplotlib\n        \n        # Set title\n        ax.set_title(f\"{item['label'] if 'label' in df.columns else ''} {item['filename']} / {item['image']} - Faces: {item['faces_mtcnn'] if 'faces_mtcnn' in df.columns else len(face_locations)} {item['faces_mtcnn_median'] if 'faces_mtcnn_median' in df.columns else ''} {item['faces'] if 'faces' in df.columns else ''}\")\n        \n        # Show plot if end of row or last item\n        if col == max_cols - 1 or idx == max_items - 1:\n            plt.show()\n        \n        idx += 1\n        if max_items > 0 and idx >= max_items:\n            break\n\nSHAPE_PREDICTOR_PATH = \"/kaggle/input/shape-predictor-68-face-landmarks-dat/shape_predictor_68_face_landmarks.dat\"\n# Replace 'image_path' with the actual column name if different\nplot_faces_boxes(faces_pd, IMAGES_FOLDER_TRAIN, SHAPE_PREDICTOR_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:13:28.217908Z","iopub.execute_input":"2024-06-21T06:13:28.218321Z","iopub.status.idle":"2024-06-21T06:13:47.377196Z","shell.execute_reply.started":"2024-06-21T06:13:28.218258Z","shell.execute_reply":"2024-06-21T06:13:47.375848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport dlib\nimport os\nimport numpy as np\nimport pandas as pd\n\n# Define the facial landmarks indexes\nFACIAL_LANDMARKS_INDEXES = {\n    \"mouth\": (48, 68),\n    \"right_eyebrow\": (17, 22),\n    \"left_eyebrow\": (22, 27),\n    \"right_eye\": (36, 42),\n    \"left_eye\": (42, 48),\n    \"nose\": (27, 36),\n    \"jaw\": (0, 17)\n}\n\n# Function to convert dlib shape object to numpy array\ndef shape_to_numpy_array(shape, dtype=\"int\"):\n    # Initialize the array with shape (68, 2)\n    coords = np.zeros((68, 2), dtype=dtype)\n    for i in range(68):\n        coords[i] = (shape.part(i).x, shape.part(i).y)\n    return coords\n\n# Function to crop and save the facial landmarks\ndef crop_and_save(image, pts, organ_name, save_folder):\n    # Create the save folder if it doesn't exist\n    os.makedirs(save_folder, exist_ok=True)\n    \n    # Determine the bounding box of the points\n    (x, y, w, h) = cv2.boundingRect(pts)\n    \n    # Extract the region of interest\n    roi = image[y:y + h, x:x + w]\n    \n    # Define the filename for saving\n    save_path = os.path.join(save_folder, f\"{organ_name}.png\")\n    \n    # Save the cropped image\n    cv2.imwrite(save_path, roi)\n\n# Function to plot faces and boxes\ndef plot_faces_boxes(df, images_folder, shape_predictor_path, max_cols=2, max_rows=6, fsize=(24, 5), max_items=12):\n    # Initialize dlib's face detector (HOG-based) and shape predictor\n    detector = dlib.get_frontal_face_detector()\n    predictor = dlib.shape_predictor(shape_predictor_path)\n\n    idx = 0\n    fig = None\n\n    # Iterate through each row in the dataframe\n    for item_idx, item in df.iterrows():\n        # Load the image\n        img_path = f\"{images_folder}/{item['filename'][:-4]}/{item['image']}\"\n        img = cv2.imread(img_path)\n        if img is None:\n            print(f\"Error: Unable to load image {img_path}\")\n            continue\n        \n        # Copy the image to draw bounding boxes\n        face_img = img.copy()\n        \n        # Calculate grid position\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0:\n            fig = plt.figure(figsize=fsize)\n        \n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        \n        # Extract face boxes and confidence scores\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols):\n            face_locations = item[c]\n            \n            if len(face_locations) > 0:\n                for face_location in face_locations:\n                    ((x, y, w, h), confidence) = face_location\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255, i*255, 0), 2)\n                    cv2.putText(face_img, f'{confidence*100:.1f}', (x, y-10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, i*255, 0), 2)\n                    \n                    # Detect facial landmarks\n                    gray = cv2.cvtColor(face_img, cv2.COLOR_BGR2GRAY)\n                    rects = detector(gray, 1)\n                    for (i, rect) in enumerate(rects):\n                        shape = predictor(gray, rect)\n                        shape_np = shape_to_numpy_array(shape)\n                        \n                        # Save each organ to the respective folder\n                        for (name, (i, j)) in FACIAL_LANDMARKS_INDEXES.items():\n                            pts = shape_np[i:j]\n                            organ_folder = os.path.join(\"/kaggle/working/facial organs\", name)\n                            organ_name = f\"{item['filename']}{item['image']}{name}_{item_idx}\"\n                            crop_and_save(face_img, pts, organ_name, organ_folder)\n                    \n                    # Draw facial landmarks on the image\n                    for (name, (i, j)) in FACIAL_LANDMARKS_INDEXES.items():\n                        pts = shape_np[i:j]\n                        cv2.rectangle(face_img, (pts[:, 0].min(), pts[:, 1].min()), (pts[:, 0].max(), pts[:, 1].max()), (255, 0, 0), 2)\n                    \n                ax.imshow(cv2.cvtColor(face_img, cv2.COLOR_BGR2RGB))  # Convert to RGB for displaying in matplotlib\n            else:\n                ax.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))  # Convert to RGB for displaying in matplotlib\n        \n        # Set title\n        ax.set_title(f\"{item['label'] if 'label' in df.columns else ''} {item['filename']} / {item['image']} - Faces: {item['faces_mtcnn'] if 'faces_mtcnn' in df.columns else len(face_locations)} {item['faces_mtcnn_median'] if 'faces_mtcnn_median' in df.columns else ''} {item['faces'] if 'faces' in df.columns else ''}\")\n        \n        # Show plot if end of row or last item\n        if col == max_cols - 1 or idx == max_items - 1:\n            plt.show()\n        \n        idx += 1\n        if max_items > 0 and idx >= max_items:\n            break\n\n# Example usage\nSHAPE_PREDICTOR_PATH = \"/kaggle/input/shape-predictor-68-face-landmarks-dat/shape_predictor_68_face_landmarks.dat\"\n # Replace with your actual dataset path\n\n# Ensure the main directory for facial organs exists\nos.makedirs(\"/kaggle/working/facial organs\", exist_ok=True)\n\n# Assuming 'faces_pd' is your dataframe with image paths and face boxes\nplot_faces_boxes(faces_pd, IMAGES_FOLDER_TRAIN, SHAPE_PREDICTOR_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:14:29.829115Z","iopub.execute_input":"2024-06-21T06:14:29.829414Z","iopub.status.idle":"2024-06-21T06:14:48.602383Z","shell.execute_reply.started":"2024-06-21T06:14:29.829372Z","shell.execute_reply":"2024-06-21T06:14:48.601199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\n\n# Define the path to your image file\nimage_path = \"/kaggle/working/facial organs/left_eyebrow/adylbeequz.mp4output_0002.pngleft_eyebrow_2.png\"\n\n# Display the image\nImage(filename=image_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:16:45.062245Z","iopub.execute_input":"2024-06-21T06:16:45.062584Z","iopub.status.idle":"2024-06-21T06:16:45.069935Z","shell.execute_reply.started":"2024-06-21T06:16:45.062539Z","shell.execute_reply":"2024-06-21T06:16:45.06893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport os\nimport cv2\nimport numpy as np\n\n# Define facial landmarks indexes\nFACIAL_LANDMARKS_INDEXES = {\n    \"mouth\": (48, 68),\n    \"right_eyebrow\": (17, 22),\n    \"left_eyebrow\": (22, 27),\n    \"right_eye\": (36, 42),\n    \"left_eye\": (42, 48),\n    \"nose\": (27, 36),\n    \"jaw\": (0, 17)\n}\n\n# Function to convert dlib shape to numpy array\ndef shape_to_numpy_array(shape, dtype=\"int\"):\n    coordinates = np.zeros((shape.num_parts, 2), dtype=dtype)\n    for i in range(0, shape.num_parts):\n        coordinates[i] = (shape.part(i).x, shape.part(i).y)\n    return coordinates\n\n# Function to extract and save facial landmarks as .jpg\ndef extract_and_save_landmarks(image_path, shape_predictor_path):\n    try:\n        # Load image\n        img = cv2.imread(image_path)\n        if img is None:\n            print(f\"Error loading image {image_path}\")\n            return None, None\n\n        gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    except Exception as e:\n        print(f\"Error processing image {image_path}: {e}\")\n        return None, None\n\n    # Initialize dlib's face detector and shape predictor\n    detector = dlib.get_frontal_face_detector()\n    predictor = dlib.shape_predictor(shape_predictor_path)\n\n    # Detect faces\n    rects = detector(gray, 1)\n    if len(rects) == 0:\n        print(f\"No faces detected in {image_path}\")\n        return None, None\n\n    # Prepare lists to store images and labels\n    images = []\n    labels = []\n\n    for (i, rect) in enumerate(rects):\n        # Predict facial landmarks\n        shape = predictor(gray, rect)\n        shape = shape_to_numpy_array(shape)\n\n        for (name, (i, j)) in FACIAL_LANDMARKS_INDEXES.items():\n            # Extract ROI for each facial landmark region\n            (x, y, w, h) = cv2.boundingRect(np.array([shape[i:j]]))\n            roi = img[y:y+h, x:x+w]\n\n            # Resize ROI to a fixed size (if needed)\n            roi = cv2.resize(roi, (128, 128))  # Adjust size as per your CNN input size\n\n            # Convert ROI to grayscale if necessary\n            if roi.shape[-1] == 3:\n                roi = cv2.cvtColor(roi, cv2.COLOR_BGR2GRAY)\n\n            # Normalize ROI pixel values (optional step, depends on your CNN requirements)\n            roi = roi.astype(np.float32) / 255.0\n\n            # Append ROI and corresponding label (name) to lists\n            images.append(roi)\n            labels.append(name)\n\n    return images, labels\n\n# Example usage\nSHAPE_PREDICTOR_PATH = \"/kaggle/input/shape-predictor-68-face-landmarks-dat/shape_predictor_68_face_landmarks.dat\"\nIMAGES_FOLDER_TRAIN = \"/kaggle/working/\"\n\n# Prepare lists to store all images and labels\nall_images = []\nall_labels = []\n\n# Iterate over all images in IMAGES_FOLDER_TRAIN\nfor filename in os.listdir(IMAGES_FOLDER_TRAIN):\n    if filename.endswith(\".jpg\") or filename.endswith(\".png\"):\n        image_path = os.path.join(IMAGES_FOLDER_TRAIN, filename)\n        print(f\"Processing {image_path}\")\n        images, labels = extract_and_save_landmarks(image_path, SHAPE_PREDICTOR_PATH)\n        if images and labels:\n            all_images.extend(images)\n            all_labels.extend(labels)\n\n# Convert lists to numpy arrays\nall_images = np.array(all_images).reshape(-1, 128, 128, 1)  # Reshape for CNN input (assuming grayscale)\nall_labels = np.array(all_labels)\n\n# Check if arrays are not empty before splitting\nif len(all_images) == 0 or len(all_labels) == 0:\n    print(\"No valid data to split. Exiting.\")\n    exit()\n\n# Shuffle the data (if needed)\nshuffle_indices = np.random.permutation(len(all_images))\nall_images = all_images[shuffle_indices]\nall_labels = all_labels[shuffle_indices]\n\n# Encode labels as integers\nlabel_encoder = LabelEncoder()\nall_labels_encoded = label_encoder.fit_transform(all_labels)\n\n# Split data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(all_images, all_labels_encoded, test_size=0.2, random_state=42)\n\n# Define the CNN model\nmodel = Sequential([\n    Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128, 1)),\n    MaxPooling2D((2, 2)),\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    Flatten(),\n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(len(FACIAL_LANDMARKS_INDEXES), activation='softmax')  # Output layer for multi-class classification\n])\n\n# Compile the model\nmodel.compile(optimizer='adam',loss='sparse_categorical_crossentropy',metrics=['accuracy'])\n\n# Print model summary\nmodel.summary()\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=10, batch_size=32, validation_data=(X_test, y_test))\n\n# Evaluate the model\ntest_loss, test_acc = model.evaluate(X_test, y_test, verbose=2)\nprint(f\"Test accuracy: {test_acc}\")\n\n# Save the model\nmodel.save(\"/kaggle/working/facial_landmarks_cnn_model.h5\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T07:28:32.495208Z","iopub.execute_input":"2024-06-21T07:28:32.495521Z","iopub.status.idle":"2024-06-21T07:28:36.086331Z","shell.execute_reply.started":"2024-06-21T07:28:32.495478Z","shell.execute_reply":"2024-06-21T07:28:36.08493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Extract frames and detect faces with landmarks from videos\ndef run_detector_on_video(videos_filename, shape_predictor_path, verbose=False):\n    if verbose: \n        print(\"Starting with batch of %d videos\" % len(videos_filename))\n    tmp_faces_pd = []\n    \n    # Initialize dlib's face detector and shape predictor\n    detector = dlib.get_frontal_face_detector()\n    predictor = dlib.shape_predictor(SHAPE_PREDICTOR_PATH)\n    \n    for file in videos_filename:\n        # Find the dump folder with images\n        dump_folder = os.path.join(IMAGES_FOLDER_TRAIN, file[:-4])\n        \n        if not os.path.exists(dump_folder):\n            if verbose:\n                print(f\"Dump folder {dump_folder} does not exist. Skipping {file}.\")\n            continue\n        \n        # List files\n        files = glob.glob(os.path.join(dump_folder, \"*\"))\n        \n        if verbose:\n            print(f\"Processing {len(files)} images from {dump_folder}\")\n        \n        DETECTORS = {\n            \"mtcnn\": detect_face_mtcnn\n            # Add more detectors if needed\n        }\n        \n        for key, detector in DETECTORS.items():\n            for image_path in files:\n                image = cv2.imread(image_path)\n                if image is None:\n                    continue\n                \n                results = detector(image)\n                faces = []\n                \n                for result in results:\n                    if key == \"mtcnn\":\n                        box = result['box']\n                        confidence = result['confidence']\n                        if confidence < 0.9:  # Filter by confidence\n                            continue\n                        \n                        x, y, w, h = box\n                        rect = dlib.rectangle(int(x), int(y), int(x + w), int(y + h))\n                        faces.append((rect, confidence))\n                    \n                    # Extract landmarks and visualize them\n                    for (rect, _) in faces:\n                        shape = predictor(image, rect)\n                        shape = shape_to_numpy_array(shape)\n                        image = visualize_facial_landmarks(image, shape)\n                \n                tmp_faces_pd.append({\n                    \"filename\": image_path,\n                    \"image\": image,\n                    \"boxes_\" + key: [face[0] for face in faces],\n                    \"faces_\" + key: len(faces)\n                })\n        \n    if len(tmp_faces_pd) > 0:\n        tmp_faces_pd = pd.DataFrame(tmp_faces_pd)\n    else:\n        tmp_faces_pd = pd.DataFrame()\n    \n    return tmp_faces_pd","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:23:05.556817Z","iopub.execute_input":"2024-06-21T06:23:05.557477Z","iopub.status.idle":"2024-06-21T06:23:05.580972Z","shell.execute_reply.started":"2024-06-21T06:23:05.557177Z","shell.execute_reply":"2024-06-21T06:23:05.580231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_detector_on_video(videos_filename, verbose=False):\n    if verbose == True: \n        print(\"Starting with batch of %d videos\" % len(videos_filename))\n    tmp_faces_pd = None\n    for file in videos_filename:\n        # Find out dump folder with images\n        dump_folder = images_folder_path + \"/\" + file[:-4]\n        # List files\n        files = glob.glob(dump_folder + \"/*\")\n        DETECTORS = {\n            \"mtcnn\": detect_face_mtcnn\n        }\n        for key, value in DETECTORS.items():\n            tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n            if tmp_faces_pd is None:\n                tmp_faces_pd = tmp_pd\n            else:\n                tmp_faces_pd = pd.concat([tmp_faces_pd, tmp_pd], axis=0)\n    return tmp_faces_pd","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:23:15.656038Z","iopub.execute_input":"2024-06-21T06:23:15.656332Z","iopub.status.idle":"2024-06-21T06:23:15.665046Z","shell.execute_reply.started":"2024-06-21T06:23:15.65629Z","shell.execute_reply":"2024-06-21T06:23:15.664174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_faces_pd[\"faces_mtcnn_avg\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmean)\nall_faces_pd[\"faces_mtcnn_median\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmedian)\nall_faces_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:23:25.990841Z","iopub.execute_input":"2024-06-21T06:23:25.991163Z","iopub.status.idle":"2024-06-21T06:23:26.020186Z","shell.execute_reply.started":"2024-06-21T06:23:25.991114Z","shell.execute_reply":"2024-06-21T06:23:26.019176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_avg\"], kde=True, ax=ax[0])\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_median\"], kde=False, ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:43:23.123502Z","iopub.execute_input":"2024-06-21T06:43:23.123875Z","iopub.status.idle":"2024-06-21T06:43:23.436091Z","shell.execute_reply.started":"2024-06-21T06:43:23.123812Z","shell.execute_reply":"2024-06-21T06:43:23.434928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(all_faces_pd[all_faces_pd[\"faces_mtcnn\"] == 3], max_items=24)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.463985Z","iopub.status.idle":"2024-06-21T06:07:13.464694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_faces_pd = pd.merge(all_faces_pd, train_pd, on=\"filename\", how=\"left\")\nclean_faces_pd.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.465837Z","iopub.status.idle":"2024-06-21T06:07:13.466383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def faces_max_item(boxes, idx1, idx2):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[idx1][idx2])[idx1][idx2]\n    return ret\n\ndef faces_max_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[1])[1]\n    return ret\n\ndef faces_min_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = min(boxes, key=lambda item: item[1])[1]\n    return ret\n\nclean_faces_pd[\"faces_max_width\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 2)) \nclean_faces_pd[\"faces_max_height\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 3))\nclean_faces_pd[\"faces_max_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_confidence(x))\nclean_faces_pd[\"faces_min_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_min_confidence(x))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:43:34.846647Z","iopub.execute_input":"2024-06-21T06:43:34.846933Z","iopub.status.idle":"2024-06-21T06:43:34.879978Z","shell.execute_reply.started":"2024-06-21T06:43:34.846891Z","shell.execute_reply":"2024-06-21T06:43:34.878864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Faces stats:\")\nprint(clean_faces_pd[[\"faces_max_width\", \"faces_max_height\", \"faces_min_conf\", \"faces_max_conf\"]].describe(percentiles=[0.01,0.05, 0.1,0.25,0.5,0.75,0.9,0.95,0.99]))\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_max_width\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_height\"], kde=True, ax=ax[1])\nplt.show()\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_min_conf\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_conf\"], kde=True, ax=ax[1])\nfig, ax = plt.subplots(figsize=(22, 3))\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_width\", y=\"faces_max_conf\", c=\"red\", ax=ax, label=\"faces_max_width\", alpha=0.5)\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_height\", y=\"faces_max_conf\", c=\"blue\", ax=d,  label=\"faces_max_height\", alpha=0.5)\nd = plt.legend(loc=\"upper right\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.469897Z","iopub.status.idle":"2024-06-21T06:07:13.470599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install imutils","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.471985Z","iopub.status.idle":"2024-06-21T06:07:13.472704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport shutil\nimport cv2\nimport pandas as pd\nimport matplotlib\nmatplotlib.use(\"Agg\")\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pickle\nfrom imutils import paths\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import VGG16\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dense\nfrom keras.layers import Input\nfrom keras.models import Model\nfrom keras.optimizers import SGD\nfrom sklearn.metrics import classification_report\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:43:55.323254Z","iopub.execute_input":"2024-06-21T06:43:55.32355Z","iopub.status.idle":"2024-06-21T06:43:55.731747Z","shell.execute_reply.started":"2024-06-21T06:43:55.323508Z","shell.execute_reply":"2024-06-21T06:43:55.731107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/working/finetuningkeras/dataset\"\n\n# define the names of the training, testing, and validation\n# directories\nTRAIN = \"training\"\nTEST = \"evaluation\"\nVAL = \"validation\"\n\nREAL = 'REAL'\nFAKE = 'FAKE'\n\n# initialize the list of class label names\nCLASSES = [\"FAKE\", \"REAL\"]\n\n\n# set the batch size when fine-tuning\nBATCH_SIZE = 32\n\ntrainEpochs = 10\nepochsFineTune = 10\nmaxVids = 5\n\n# set the path to the serialized model after training\nMODEL_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"Deepfake.model\"])\n\n# define the path to the output training history plots\nUNFROZEN_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"unfrozen.png\"])\nWARMUP_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"warmup.png\"])\n\nfile = '/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json'\nimg_path = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ndata_path = '/kaggle/working/finetuningkeras/real_fake'\ndir_fake_frames = '/kaggle/working/FAKE_frames'\ndir_real_frames = '/kaggle/working/REAL_frames'\ndir_output = '/kaggle/working/finetuningkeras/output'\n\ndir_data_path_real = os.path.join(data_path, REAL)\ndir_data_path_fake = os.path.join(data_path, FAKE)\n\ndir_train_real = os.path.join(BASE_PATH, TRAIN, REAL)\ndir_train_fake = os.path.join(BASE_PATH, TRAIN, FAKE)\ndir_valid_real = os.path.join(BASE_PATH, VAL, REAL)\ndir_valid_fake = os.path.join(BASE_PATH, VAL, FAKE)\ndir_test_real = os.path.join(BASE_PATH, TEST, REAL)\ndir_test_fake = os.path.join(BASE_PATH, TEST, FAKE)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:43:48.256279Z","iopub.execute_input":"2024-06-21T06:43:48.256628Z","iopub.status.idle":"2024-06-21T06:43:48.269497Z","shell.execute_reply.started":"2024-06-21T06:43:48.256567Z","shell.execute_reply":"2024-06-21T06:43:48.268734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = '/kaggle/working/finetuningkeras/real_fake/FAKE'\noutput_dir = '/kaggle/working/FAKE_frames/'\ndef explode_frames(input_dir, output_dir, maxN):\n\n    mp4_filenames = [f for f in os.listdir(input_dir) if f.endswith('.mp4')]\n    n = 0\n    \n    for mp4fn in mp4_filenames:\n        \n        if(n < maxN):\n            n += 1 \n            mp4fp = os.path.join(input_dir, mp4fn)\n            cam = cv2.VideoCapture(mp4fp) \n            if(cam.isOpened()):\n                print('Processing file #'+ str(n) + ' (' + mp4fn + ')...')\n            else: \n                print('Problem opening file #'+ str(n) + ' (' + mp4fn + ')...')\n                continue \n            \n            nframe = 0\n            while(True): #continue until ret = False then break\n                nframe += 1\n                ret,frame = cam.read()\n                \n                if ret: \n                    # if video is still left continue creating images \n                    out_filename = os.path.splitext(mp4fn)[0]+  '_frame' + str(nframe) + '.jpg'\n                    out_filepath =  os.path.join(output_dir, out_filename)\n                    \n                    # writing the extracted images \n                    cv2.imwrite(out_filepath, frame) \n                else: \n                    break\n\n            # Release all space and windows once done\n            print(' - created ' + str(nframe-1) + ' images') # -1 bc count incremented before exit\n            cam.release() \n            cv2.destroyAllWindows()\n            \n        else: \n            break\n            \n\"\"\"\nDistribute files/images from a source directory into training, validation, and testing directories. \nsrc_dir = source/input directory\ntrain_dir, val_dir, test_dir = target training/validation/testing directory\nvalperc = fraction of dataset to use for validation (0-1)\ntestperc = fraction of dataset to use for testing (0-1)\n\"\"\"\n            \ndef trainvaltest_split(src_dir, train_dir, val_dir, test_dir, valperc = 0.15, testperc = 0.15):\n    \n    filenames = os.listdir(src_dir) #get all filenames in random order\n    np.random.shuffle(filenames)\n    \n    n = len(filenames)\n    split1 = int(n*(1 - (valperc + testperc)))\n    split2 = int(n*(1 - (testperc)))\n    \n    fn_train, fn_val, fn_test = np.split(np.array(filenames), [split1, split2])\n    \n    fn_lists = [fn_train, fn_val, fn_test]\n    targetdirs = [train_dir, val_dir, test_dir]\n    \n    print('Total images: ', n)\n    print('Training: ', len(fn_train))\n    print('Validation: ', len(fn_val))\n    print('Testing: ', len(fn_test))\n    \n    all_fp = [os.path.join(src_dir, fn) for fn in filenames]\n    \n    #move files\n    for i, fn_list in enumerate(fn_lists):\n        for fn in fn_list: \n            target_dir = targetdirs[i]\n            fp_from = os.path.join(src_dir, fn)\n            fp_to = os.path.join(target_dir, fn)\n            \n            shutil.move(fp_from, fp_to)\n\n            \n\"\"\"\nConstruct a plot that plots and saves the training history\n\"\"\"           \ndef plot_training(H, N, plotPath):\n\tplt.style.use(\"ggplot\")\n\tplt.figure()\n\tplt.plot(np.arange(0, N), H.history[\"loss\"], label=\"train_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"val_loss\"], label=\"val_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"accuracy\"], label=\"train_acc\")\n\tplt.plot(np.arange(0, N), H.history[\"val_accuracy\"], label=\"val_acc\")\n\tplt.title(\"Training Loss and Accuracy\")\n\tplt.xlabel(\"Epoch #\")\n\tplt.ylabel(\"Loss/Accuracy\")\n\tplt.legend(loc=\"lower left\")\n\tplt.savefig(plotPath)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:44:00.940957Z","iopub.execute_input":"2024-06-21T06:44:00.941299Z","iopub.status.idle":"2024-06-21T06:44:00.967726Z","shell.execute_reply.started":"2024-06-21T06:44:00.941252Z","shell.execute_reply":"2024-06-21T06:44:00.966964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nos.makedirs(dir_train_real, exist_ok = True)\nos.makedirs(dir_train_fake, exist_ok = True)\nos.makedirs(dir_valid_real, exist_ok = True)\nos.makedirs(dir_valid_fake, exist_ok = True)\nos.makedirs(dir_test_real, exist_ok = True)\nos.makedirs(dir_test_fake, exist_ok = True)\n\nos.makedirs(dir_data_path_real, exist_ok = True)\nos.makedirs(dir_data_path_fake, exist_ok = True)\nos.makedirs(dir_fake_frames, exist_ok = True) \nos.makedirs(dir_real_frames, exist_ok = True) \nos.makedirs(dir_output, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:44:20.578142Z","iopub.execute_input":"2024-06-21T06:44:20.578436Z","iopub.status.idle":"2024-06-21T06:44:20.587071Z","shell.execute_reply.started":"2024-06-21T06:44:20.578386Z","shell.execute_reply":"2024-06-21T06:44:20.586164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_json(file)\ndf = df.T\n\n# %% [code]\nlabel = df[['label']]","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:44:24.651219Z","iopub.execute_input":"2024-06-21T06:44:24.651596Z","iopub.status.idle":"2024-06-21T06:44:24.822199Z","shell.execute_reply.started":"2024-06-21T06:44:24.651473Z","shell.execute_reply":"2024-06-21T06:44:24.821512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fn, row in label.iterrows():\n    src = os.path.join(img_path, fn)\n    dest = os.path.join(data_path, row['label'], fn)\n    shutil.copy(src, dest)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:44:36.668588Z","iopub.execute_input":"2024-06-21T06:44:36.668931Z","iopub.status.idle":"2024-06-21T06:44:40.688193Z","shell.execute_reply.started":"2024-06-21T06:44:36.668876Z","shell.execute_reply":"2024-06-21T06:44:40.687327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_fake, dir_fake_frames, maxN= maxVids)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:44:46.663857Z","iopub.execute_input":"2024-06-21T06:44:46.664193Z","iopub.status.idle":"2024-06-21T06:45:39.815333Z","shell.execute_reply.started":"2024-06-21T06:44:46.664133Z","shell.execute_reply":"2024-06-21T06:45:39.814565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_real, dir_real_frames, maxN= maxVids)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:48:31.128045Z","iopub.execute_input":"2024-06-21T06:48:31.128352Z","iopub.status.idle":"2024-06-21T06:49:24.619716Z","shell.execute_reply.started":"2024-06-21T06:48:31.128311Z","shell.execute_reply":"2024-06-21T06:49:24.618867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_fake_frames,\n                   train_dir = dir_train_fake, \n                   val_dir = dir_valid_fake, \n                   test_dir = dir_test_fake)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:29.637787Z","iopub.execute_input":"2024-06-21T06:49:29.638116Z","iopub.status.idle":"2024-06-21T06:49:29.699646Z","shell.execute_reply.started":"2024-06-21T06:49:29.638065Z","shell.execute_reply":"2024-06-21T06:49:29.698796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_real_frames,\n                   train_dir = dir_train_real, \n                   val_dir = dir_valid_real, \n                   test_dir = dir_test_real)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:33.139057Z","iopub.execute_input":"2024-06-21T06:49:33.139538Z","iopub.status.idle":"2024-06-21T06:49:33.199585Z","shell.execute_reply.started":"2024-06-21T06:49:33.139313Z","shell.execute_reply":"2024-06-21T06:49:33.198856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrainAug = ImageDataGenerator(\n\trotation_range=30,\n\tzoom_range=0.15,\n\twidth_shift_range=0.2,\n\theight_shift_range=0.2,\n\tshear_range=0.15,\n\thorizontal_flip=True,\n\tfill_mode=\"nearest\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:37.715336Z","iopub.execute_input":"2024-06-21T06:49:37.715645Z","iopub.status.idle":"2024-06-21T06:49:37.720423Z","shell.execute_reply.started":"2024-06-21T06:49:37.715604Z","shell.execute_reply":"2024-06-21T06:49:37.719554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valAug = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:42.541874Z","iopub.execute_input":"2024-06-21T06:49:42.542164Z","iopub.status.idle":"2024-06-21T06:49:42.546809Z","shell.execute_reply.started":"2024-06-21T06:49:42.542123Z","shell.execute_reply":"2024-06-21T06:49:42.545849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = np.array([123.68, 116.779, 103.939], dtype=\"float32\")\ntrainAug.mean = mean\nvalAug.mean = mean","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:46.087693Z","iopub.execute_input":"2024-06-21T06:49:46.087968Z","iopub.status.idle":"2024-06-21T06:49:46.092314Z","shell.execute_reply.started":"2024-06-21T06:49:46.087928Z","shell.execute_reply":"2024-06-21T06:49:46.091598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainPath = os.path.join(BASE_PATH, TRAIN)\ntrainGen = trainAug.flow_from_directory(\n\ttrainPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=True,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the validation generator\nvalPath = os.path.join(BASE_PATH, VAL)\nvalGen = valAug.flow_from_directory(\n\tvalPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the testing generator\ntestPath = os.path.join(BASE_PATH, TEST)\ntestGen = valAug.flow_from_directory(\n\ttestPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:51.64778Z","iopub.execute_input":"2024-06-21T06:49:51.648082Z","iopub.status.idle":"2024-06-21T06:49:51.96951Z","shell.execute_reply.started":"2024-06-21T06:49:51.648037Z","shell.execute_reply":"2024-06-21T06:49:51.9686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseModel = VGG16(weights=\"imagenet\", include_top=False,\n\tinput_tensor=Input(shape=(224, 224, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:49:56.825115Z","iopub.execute_input":"2024-06-21T06:49:56.825459Z","iopub.status.idle":"2024-06-21T06:49:57.648822Z","shell.execute_reply.started":"2024-06-21T06:49:56.825395Z","shell.execute_reply":"2024-06-21T06:49:57.648088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headModel = baseModel.output\nheadModel = Flatten(name=\"flatten\")(headModel)\nheadModel = Dense(512, activation=\"relu\")(headModel)\nheadModel = Dropout(0.5)(headModel)\nheadModel = Dense(len(CLASSES), activation=\"softmax\")(headModel)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:50:14.340677Z","iopub.execute_input":"2024-06-21T06:50:14.340975Z","iopub.status.idle":"2024-06-21T06:50:14.384738Z","shell.execute_reply.started":"2024-06-21T06:50:14.340932Z","shell.execute_reply":"2024-06-21T06:50:14.384037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=baseModel.input, outputs=headModel)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:50:18.251313Z","iopub.execute_input":"2024-06-21T06:50:18.251782Z","iopub.status.idle":"2024-06-21T06:50:18.256851Z","shell.execute_reply.started":"2024-06-21T06:50:18.251591Z","shell.execute_reply":"2024-06-21T06:50:18.256129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers:\n\tlayer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:50:25.701699Z","iopub.execute_input":"2024-06-21T06:50:25.701995Z","iopub.status.idle":"2024-06-21T06:50:25.705766Z","shell.execute_reply.started":"2024-06-21T06:50:25.70195Z","shell.execute_reply":"2024-06-21T06:50:25.705074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:50:28.804366Z","iopub.execute_input":"2024-06-21T06:50:28.804664Z","iopub.status.idle":"2024-06-21T06:50:28.854898Z","shell.execute_reply.started":"2024-06-21T06:50:28.804615Z","shell.execute_reply":"2024-06-21T06:50:28.85409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"totalTrain = len(list(paths.list_images(trainPath)))\ntotalVal = len(list(paths.list_images(valPath)))\ntotalTest = len(list(paths.list_images(testPath)))\n\nprint(\"[INFO] training head...\")\nH = model.fit(\n    trainGen,\n    steps_per_epoch=totalTrain // BATCH_SIZE,\n    validation_data=valGen,\n    validation_steps=totalVal // BATCH_SIZE,\n    epochs=trainEpochs)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:51:54.935345Z","iopub.execute_input":"2024-06-21T06:51:54.935652Z","iopub.status.idle":"2024-06-21T07:05:16.044278Z","shell.execute_reply.started":"2024-06-21T06:51:54.93561Z","shell.execute_reply":"2024-06-21T07:05:16.042599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network head...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\n\n\nplot_training(H, trainEpochs, WARMUP_PLOT_PATH)\nplt.show()  ","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.512429Z","iopub.status.idle":"2024-06-21T06:07:13.512903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.514352Z","iopub.status.idle":"2024-06-21T06:07:13.515144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns ","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.516205Z","iopub.status.idle":"2024-06-21T06:07:13.516705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming testGen and model are defined\ntestGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\ny_true = testGen.classes\n\n# Compute ROC curve for the positive class\nfpr, tpr, _ = roc_curve(y_true, y_pred_probs[:, 1])\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=1, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.518087Z","iopub.status.idle":"2024-06-21T06:07:13.518583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.51985Z","iopub.status.idle":"2024-06-21T06:07:13.520538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.521942Z","iopub.status.idle":"2024-06-21T06:07:13.522698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.524191Z","iopub.status.idle":"2024-06-21T06:07:13.524693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainGen.reset()\nvalGen.reset()","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.525878Z","iopub.status.idle":"2024-06-21T06:07:13.526382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers[15:]:\n\tlayer.trainable = True\n\n# loop over the layers in the model and show which ones are trainable\n# or not\nfor layer in baseModel.layers:\n\tprint(\"{}: {}\".format(layer, layer.trainable))\n\n# for the changes to the model to take affect we need to recompile\n# the model, this time using SGD with a *very* small learning rate\nprint(\"[INFO] re-compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.527523Z","iopub.status.idle":"2024-06-21T06:07:13.528027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"H = model.fit_generator(\n\ttrainGen,\n\tsteps_per_epoch=totalTrain // BATCH_SIZE,\n\tvalidation_data=valGen,\n\tvalidation_steps=totalVal // BATCH_SIZE,\n\tepochs= epochsFineTune)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.529428Z","iopub.status.idle":"2024-06-21T06:07:13.529958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\nplot_training(H, epochsFineTune, UNFROZEN_PLOT_PATH)\n\n# serialize the model to disk\nprint(\"[INFO] serializing network...\")\nmodel.save(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.531308Z","iopub.status.idle":"2024-06-21T06:07:13.531798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.533143Z","iopub.status.idle":"2024-06-21T06:07:13.533714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.534929Z","iopub.status.idle":"2024-06-21T06:07:13.535669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.537147Z","iopub.status.idle":"2024-06-21T06:07:13.537651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wave\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.539794Z","iopub.status.idle":"2024-06-21T06:07:13.540303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"margot_robbie_speech = wave.open('/kaggle/input/deep-voice-deepfake-voice-recognition/KAGGLE/AUDIO/REAL/margot-original.wav', 'rb')\nprint(margot_robbie_speech)\n\nsample_freq = margot_robbie_speech.getframerate()\nn_samples = margot_robbie_speech.getnframes()\nt_audio = n_samples/sample_freq\nn_channels = margot_robbie_speech.getnchannels()\n\nprint(\"The samping rate of the audio file is \" + str(sample_freq) + \"Hz, or \" + str(sample_freq/1000) + \"kHz\")\nprint(\"The audio contains a total of \" + str(n_samples) + \" frames or samples\")\nprint(\"The length of the audio file is \" + str(t_audio) + \" seconds\")\nprint(\"The audio file has \" + str(n_channels) + \" channels.\\n\")\n\n\n\nsignal_wave = margot_robbie_speech.readframes(n_samples)\nsignal_array = np.frombuffer(signal_wave, dtype=np.int16)\nprint(\"The signal contains a total of \" + str(signal_array.shape[0]) + \" samples.\")\nprint(\"If this value is greater than \" + str(n_samples) + \" it is due to there being multiple channels\")\nprint(\"E.g. - Samples * Channels = \" + str(n_samples*n_channels))\n\n# Split the channels\nl_channel = signal_array[0::2]\nr_channel = signal_array[1::2]","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.541497Z","iopub.status.idle":"2024-06-21T06:07:13.542124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(\"/kaggle/input/deep-voice-deepfake-voice-recognition/KAGGLE/DATASET-balanced.csv\")\n\nX = df.iloc[:,:-1]\ny = df.iloc[:,-1]\n\nprint(X.head(10))\nprint(y.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.543233Z","iopub.status.idle":"2024-06-21T06:07:13.543897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nlb = preprocessing.LabelBinarizer()\nlb.fit(y)\ny = lb.transform(y)\ny = y.ravel()\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.545045Z","iopub.status.idle":"2024-06-21T06:07:13.545596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier(n_estimators=80, random_state=1)\n\nfrom sklearn.model_selection import KFold\nkf = KFold(n_splits=10,  shuffle=True, random_state=1)\n\nprint(model)\nprint(\"KFold splits: \" + str(kf.get_n_splits(X)))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.546637Z","iopub.status.idle":"2024-06-21T06:07:13.547285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\n\nimport numpy as np\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, matthews_corrcoef, roc_auc_score\n\nacc_score = []\nprec_score = []\nrec_score = []\nf1s = []\nMCCs = []\nROCareas = []\n\nstart = time.time()\nfor train_index , test_index in kf.split(X):\n    X_train , X_test = X.iloc[train_index,:],X.iloc[test_index,:]\n    y_train , y_test = y[train_index] , y[test_index]\n     \n    model.fit(X_train,y_train)\n    pred_values = model.predict(X_test)\n    acc = accuracy_score(pred_values , y_test)\n    acc_score.append(acc)\n    \n    prec = precision_score(y_test , pred_values, average=\"binary\", pos_label=1)\n    prec_score.append(prec)\n    \n    rec = recall_score(y_test , pred_values, average=\"binary\", pos_label=1)\n    rec_score.append(rec)\n    \n    f1 = f1_score(y_test , pred_values, average=\"binary\", pos_label=1)\n    f1s.append(f1)\n    \n    mcc = matthews_corrcoef(y_test , pred_values)\n    MCCs.append(mcc)   \n    \n    roc = roc_auc_score(y_test , pred_values)\n    ROCareas.append(roc)\nend = time.time()\ntimeTaken = (end - start)\nprint(\"Model trained in: \" + str( round(timeTaken, 2) ) + \" seconds.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.548407Z","iopub.status.idle":"2024-06-21T06:07:13.549066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Mean results and (std.):\\n\")\nprint(\"Accuracy: \" + str( round(np.mean(acc_score)*100, 3) ) + \"% (\" + str( round(np.std(acc_score)*100, 3) ) + \")\\n\")\nprint(\"Precision: \" + str( round(np.mean(prec_score), 3) ) + \" (\" + str( round(np.std(prec_score), 3) ) + \")\")\nprint(\"Recall: \" + str( round(np.mean(rec_score), 3) ) + \" (\" + str( round(np.std(rec_score), 3) ) + \")\")\nprint(\"F1-Score: \" + str( round(np.mean(f1s), 3) ) + \" (\" + str( round(np.std(f1s), 3) ) + \")\")\nprint(\"MCC: \" + str( round(np.mean(MCCs), 3) ) + \" (\" + str( round(np.std(MCCs), 3) ) + \")\")\nprint(\"ROC AUC: \" + str( round(np.mean(ROCareas), 3) ) + \" (\" + str( round(np.std(ROCareas), 3) ) + \")\")","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.550174Z","iopub.status.idle":"2024-06-21T06:07:13.550742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport time\nimport matplotlib.pyplot as plt\n\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import KFold, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, matthews_corrcoef, roc_auc_score\n\n# Load dataset\ndf = pd.read_csv(\"/kaggle/input/deep-voice-deepfake-voice-recognition/KAGGLE/DATASET-balanced.csv\")\n\n# Data Exploration\nprint(df.head())\nprint(df.describe())\nprint(df.info())\n\n# Splitting the dataset into features and target\nX = df.iloc[:,:-1]\ny = df.iloc[:,-1]\n\n# Label Binarization\nlb = preprocessing.LabelBinarizer()\nlb.fit(y)\ny = lb.transform(y)\ny = y.ravel()\n\n# Feature scaling (Optional)\n# from sklearn.preprocessing import StandardScaler\n# scaler = StandardScaler()\n# X = scaler.fit_transform(X)\n\n# Initialize model and KFold\nmodel = RandomForestClassifier(n_estimators=80, random_state=1)\nkf = KFold(n_splits=10,  shuffle=True, random_state=1)\n\n# Metrics to keep track\nacc_score = []\nprec_score = []\nrec_score = []\nf1s = []\nMCCs = []\nROCareas = []\n\n# Model training and evaluation\nstart = time.time()\nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X.iloc[train_index,:], X.iloc[test_index,:]\n    y_train, y_test = y[train_index], y[test_index]\n    \n    model.fit(X_train, y_train)\n    pred_values = model.predict(X_test)\n    \n    acc_score.append(accuracy_score(y_test, pred_values))\n    prec_score.append(precision_score(y_test, pred_values))\n    rec_score.append(recall_score(y_test, pred_values))\n    f1s.append(f1_score(y_test, pred_values))\n    MCCs.append(matthews_corrcoef(y_test, pred_values))\n    ROCareas.append(roc_auc_score(y_test, pred_values))\n\nend = time.time()\n\n# Display Results\nprint(f\"Model trained in {round(end - start, 2)} seconds.\")\nprint(\"Mean results and (std.):\\n\")\nprint(f\"Accuracy: {round(np.mean(acc_score)*100, 3)}% ({round(np.std(acc_score)*100, 3)})\")\nprint(f\"Precision: {round(np.mean(prec_score), 3)} ({round(np.std(prec_score), 3)})\")\nprint(f\"Recall: {round(np.mean(rec_score), 3)} ({round(np.std(rec_score), 3)})\")\nprint(f\"F1-Score: {round(np.mean(f1s), 3)} ({round(np.std(f1s), 3)})\")\nprint(f\"MCC: {round(np.mean(MCCs), 3)} ({round(np.std(MCCs), 3)})\")\nprint(f\"ROC AUC: {round(np.mean(ROCareas), 3)} ({round(np.std(ROCareas), 3)})\")\n\n# Performance Visualization\nplt.figure(figsize=(12, 6))\nplt.plot(acc_score, label='Accuracy')\nplt.plot(prec_score, label='Precision')\nplt.plot(rec_score, label='Recall')\nplt.plot(f1s, label='F1 Score')\nplt.plot(MCCs, label='MCC')\nplt.plot(ROCareas, label='ROC AUC')\nplt.title('Performance metrics across folds')\nplt.xlabel('Fold')\nplt.ylabel('Metric Value')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-21T06:07:13.552045Z","iopub.status.idle":"2024-06-21T06:07:13.552676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}