{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":18147,"sourceType":"datasetVersion","datasetId":13405},{"sourceId":842050,"sourceType":"datasetVersion","datasetId":444558},{"sourceId":893807,"sourceType":"datasetVersion","datasetId":451078},{"sourceId":6358196,"sourceType":"datasetVersion","datasetId":3579787}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Bu kodlar, Python programında NumPy ve Pandas kütüphanelerini kullanılabilir hale getirir.\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-31T01:15:43.262761Z","iopub.execute_input":"2024-03-31T01:15:43.263055Z","iopub.status.idle":"2024-03-31T01:15:43.798094Z","shell.execute_reply.started":"2024-03-31T01:15:43.263009Z","shell.execute_reply":"2024-03-31T01:15:43.797196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Python'da Facenet modelini kullanmak için gerekli olan facenet-pytorch kütüphanesinin kurulumunu yapar.\n!pip install facenet-pytorch","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:15:48.880827Z","iopub.execute_input":"2024-03-31T01:15:48.881172Z","iopub.status.idle":"2024-03-31T01:15:56.84204Z","shell.execute_reply.started":"2024-03-31T01:15:48.881111Z","shell.execute_reply":"2024-03-31T01:15:56.841274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\nimport numpy as np\nimport pandas as pd\nimport matplotlib\nimport seaborn as sns\nimport torch\nimport matplotlib.pyplot as plt\n# from tqdm import tqdm_notebook\n%matplotlib inline \n# from google.colab.patches import cv2_imshow\nfrom IPython.display import HTML #imports to play videos\nfrom base64 import b64encode \nimport cv2 as cv\nfrom skimage.measure import compare_ssim\nimport glob\nimport time\nfrom PIL import Image\nfrom facenet_pytorch import MTCNN, InceptionResnetV1, extract_face\nfrom tqdm import tqdm\n\nimport math\nimport pickle\nfrom functools import partial\nfrom collections import defaultdict\n\nfrom PIL import Image\nfrom glob import glob\n\nimport cv2\nimport skimage.measure\nimport albumentations as A\nfrom tqdm.notebook import tqdm \nfrom albumentations.pytorch import ToTensor \n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\nfrom torchvision.models.video import mc3_18, r2plus1d_18\n\nfrom facenet_pytorch import MTCNN\n#Bu kod parçacığı, çeşitli görevler için gerekli olan kütüphaneleri ve modülleri içerir,\n#örneğin veri manipülasyonu, görselleştirme, derin öğrenme ve yüz tanıma gibi işlemler için.\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T11:59:13.765049Z","iopub.execute_input":"2024-03-31T11:59:13.765451Z","iopub.status.idle":"2024-03-31T11:59:17.563988Z","shell.execute_reply.started":"2024-03-31T11:59:13.765376Z","shell.execute_reply":"2024-03-31T11:59:17.562935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = \"../input/deepfake-detection-challenge\" \nTRAIN_SAMPLE_FOLDER = \"train_sample_videos\"\nTEST_FOLDER = \"test_videos\"\n#Bu kod parçacığında, belirli bir veri kümesine ve bu veri kümesinin içindeki klasörlere ilişkin dosya yolları tanımlanmıştır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:40.960277Z","iopub.execute_input":"2024-03-31T01:16:40.960655Z","iopub.status.idle":"2024-03-31T01:16:40.965432Z","shell.execute_reply.started":"2024-03-31T01:16:40.960598Z","shell.execute_reply":"2024-03-31T01:16:40.964405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '../input/haarcascades'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")  \n#Bu kod parçacığı, yüz tespiti için kullanılan kaynak dosyaların bulunduğu klasörün yolu belirlenir ve bu klasördeki dosyaların listesi yazdırılır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:42.910234Z","iopub.execute_input":"2024-03-31T01:16:42.91052Z","iopub.status.idle":"2024-03-31T01:16:42.928757Z","shell.execute_reply.started":"2024-03-31T01:16:42.910479Z","shell.execute_reply":"2024-03-31T01:16:42.927973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\") \n#Bu kod parçacığı, eğitim verilerinin bulunduğu klasördeki dosyaların uzantılarını kontrol eder ve eşsiz uzantıları bir listeye ekler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:44.704479Z","iopub.execute_input":"2024-03-31T01:16:44.704775Z","iopub.status.idle":"2024-03-31T01:16:44.853102Z","shell.execute_reply.started":"2024-03-31T01:16:44.704732Z","shell.execute_reply":"2024-03-31T01:16:44.852231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")\n#Bu kod parçacığı, eğitim verilerinin bulunduğu klasördeki dosyaların uzantılarını kontrol eder ve eşsiz uzantıları bir listeye ekler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:46.893598Z","iopub.execute_input":"2024-03-31T01:16:46.893892Z","iopub.status.idle":"2024-03-31T01:16:47.11778Z","shell.execute_reply.started":"2024-03-31T01:16:46.89385Z","shell.execute_reply":"2024-03-31T01:16:47.117046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")\n#reading the json file\ndef get_meta_from_json(path):\n    df = pd.read_json(os.path.join(DATA_FOLDER, path, json_file))\n    df = df.T\n    return df\n\nmeta_train_df = get_meta_from_json(TRAIN_SAMPLE_FOLDER)\nmeta_train_df.head()\n#Bu kod parçacığı, eğitim verilerinin bulunduğu klasördeki JSON dosyasını bulur ve içeriğini okur. ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:49.358799Z","iopub.execute_input":"2024-03-31T01:16:49.359094Z","iopub.status.idle":"2024-03-31T01:16:49.777292Z","shell.execute_reply.started":"2024-03-31T01:16:49.359041Z","shell.execute_reply":"2024-03-31T01:16:49.776581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))\n#Bu kod parçacığı, veri setindeki eksik verileri hesaplamak için bir fonksiyon tanımlar.\n#Fonksiyon, bir DataFrame'i alır ve her sütundaki eksik değerlerin toplam sayısını ve yüzdesini hesaplar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:52.61882Z","iopub.execute_input":"2024-03-31T01:16:52.619147Z","iopub.status.idle":"2024-03-31T01:16:52.626511Z","shell.execute_reply.started":"2024-03-31T01:16:52.619098Z","shell.execute_reply":"2024-03-31T01:16:52.625428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df)\n#Bu kod, meta_train_df adlı DataFrame'in eksik veri istatistiklerini hesaplamak için missing_data fonksiyonunu çağırır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:55.10666Z","iopub.execute_input":"2024-03-31T01:16:55.106952Z","iopub.status.idle":"2024-03-31T01:16:55.150984Z","shell.execute_reply.started":"2024-03-31T01:16:55.106909Z","shell.execute_reply":"2024-03-31T01:16:55.150207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df.loc[meta_train_df.label == 'REAL'])\n#Bu kod, meta_train_df DataFrame'indeki etiketi 'REAL' olan örneklerin eksik veri istatistiklerini hesaplamak için missing_data fonksiyonunu çağırır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:16:57.173856Z","iopub.execute_input":"2024-03-31T01:16:57.174234Z","iopub.status.idle":"2024-03-31T01:16:57.193143Z","shell.execute_reply.started":"2024-03-31T01:16:57.174176Z","shell.execute_reply":"2024-03-31T01:16:57.192006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Totals']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique() #collect all unique instances\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))\n#Bu kod parçacığı, her sütundaki benzersiz değerlerin sayısını hesaplamak için bir fonksiyon tanımlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:02.253509Z","iopub.execute_input":"2024-03-31T01:17:02.253848Z","iopub.status.idle":"2024-03-31T01:17:02.259857Z","shell.execute_reply.started":"2024-03-31T01:17:02.253788Z","shell.execute_reply":"2024-03-31T01:17:02.259026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(meta_train_df)\n#Bu kod, meta_train_df DataFrame'indeki her bir sütundaki benzersiz değerlerin sayısını hesaplamak için unique_values fonksiyonunu çağırır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:04.441887Z","iopub.execute_input":"2024-03-31T01:17:04.442205Z","iopub.status.idle":"2024-03-31T01:17:04.458307Z","shell.execute_reply.started":"2024-03-31T01:17:04.442158Z","shell.execute_reply":"2024-03-31T01:17:04.457368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_frequent_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    items = []\n    vals = []\n    for col in data.columns:\n        itm = data[col].value_counts().index[0]\n        val = data[col].value_counts().values[0]\n        items.append(itm)\n        vals.append(val)\n    tt['Most frequent item'] = items\n    tt['Frequence'] = vals\n    tt['Percent from total'] = np.round(vals / total * 100, 3)\n    return(np.transpose(tt))\n#Bu kod parçacığı, her sütundaki en sık görülen değeri ve bu değerin toplam değerler içindeki yüzdesini hesaplamak için bir fonksiyon tanımlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:06.435182Z","iopub.execute_input":"2024-03-31T01:17:06.435626Z","iopub.status.idle":"2024-03-31T01:17:06.443434Z","shell.execute_reply.started":"2024-03-31T01:17:06.435436Z","shell.execute_reply":"2024-03-31T01:17:06.442742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df)\n#Bu kod, meta_train_df DataFrame'indeki her sütun için en sık görülen değeri, bu değerin frekansını ve toplam değerler içindeki yüzdesini hesaplar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:09.989804Z","iopub.execute_input":"2024-03-31T01:17:09.990105Z","iopub.status.idle":"2024-03-31T01:17:10.0139Z","shell.execute_reply.started":"2024-03-31T01:17:09.990048Z","shell.execute_reply":"2024-03-31T01:17:10.013199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df.loc[meta_train_df.label == 'FAKE'])\n#Bu kod parçacığı, etiketi 'FAKE' olan örneklerin meta_train_df DataFrame'indeki her sütun için en sık görülen değerini,\n#bu değerin frekansını ve toplam değerler içindeki yüzdesini hesaplar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:12.2407Z","iopub.execute_input":"2024-03-31T01:17:12.241038Z","iopub.status.idle":"2024-03-31T01:17:12.268896Z","shell.execute_reply.started":"2024-03-31T01:17:12.240974Z","shell.execute_reply":"2024-03-31T01:17:12.26785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n  '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n  '''  \n  f, ax = plt.subplots(1,1, figsize=(4*size,4))\n  total = float(len(df))\n  g =  sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n  g.set_title(\"Number and percentage of {}\".format(title)) \n  if(size > 2):\n    plt.xticks(rotation=90, size=8)\n  for p in ax.patches:\n     height = p.get_height()\n     ax.text(p.get_x()+ p.get_width()/2.,height + 3,'{:1.2f}%'.format(100*height/total),ha=\"center\")\n\n  plt.show()\n#Bu kod parçacığı, bir özniteliğin sınıf dağılımını görselleştirmek için bir fonksiyon tanımlar. ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:14.372963Z","iopub.execute_input":"2024-03-31T01:17:14.373264Z","iopub.status.idle":"2024-03-31T01:17:14.383641Z","shell.execute_reply.started":"2024-03-31T01:17:14.37322Z","shell.execute_reply":"2024-03-31T01:17:14.382624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('split','split(train)',meta_train_df)\n#Bu kod parçacığı, 'split' adlı özniteliğin sınıf dağılımını görselleştirmek için plot_count fonksiyonunu çağırır.\n#Bu işlem, 'split' özniteliğinin değerlerinin sınıf dağılımını görsel olarak analiz etmek için kullanılır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:17.005949Z","iopub.execute_input":"2024-03-31T01:17:17.006294Z","iopub.status.idle":"2024-03-31T01:17:17.301981Z","shell.execute_reply.started":"2024-03-31T01:17:17.006243Z","shell.execute_reply":"2024-03-31T01:17:17.300886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('label','label(train)',meta_train_df)\n#Bu kod parçacığı, 'label' adlı özniteliğin sınıf dağılımını görselleştirmek için plot_count fonksiyonunu çağırır.\n#Bu işlem, 'label' özniteliğinin değerlerinin sınıf dağılımını görsel olarak analiz etmek için kullanılır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:19.666083Z","iopub.execute_input":"2024-03-31T01:17:19.666375Z","iopub.status.idle":"2024-03-31T01:17:19.879619Z","shell.execute_reply.started":"2024-03-31T01:17:19.666333Z","shell.execute_reply":"2024-03-31T01:17:19.878472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = np.array(list(meta_train_df.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")\n#Bu kod parçacığı, 'meta_train_df' DataFrame'indeki indeksleri (metadata), 'train_list' listesindeki dosyaların adlarıyla karşılaştırarak\n#bazı karşılaştırma işlemleri yapar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:28.876206Z","iopub.execute_input":"2024-03-31T01:17:28.876518Z","iopub.status.idle":"2024-03-31T01:17:28.885256Z","shell.execute_reply.started":"2024-03-31T01:17:28.876466Z","shell.execute_reply":"2024-03-31T01:17:28.884463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='FAKE'].sample(3).index)\nfake_train_sample_video\n#Bu kod parçacığı, 'meta_train_df' DataFrame'inde etiketi 'FAKE' olan örneklerden rastgele seçilen 3 örneğin adlarını içeren bir liste oluşturur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:31.432921Z","iopub.execute_input":"2024-03-31T01:17:31.433354Z","iopub.status.idle":"2024-03-31T01:17:31.442221Z","shell.execute_reply.started":"2024-03-31T01:17:31.433291Z","shell.execute_reply":"2024-03-31T01:17:31.441468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_img = cv.VideoCapture(video_path)\n    ret, frame = capture_img.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)\n    #Bu kod parçacığı, bir video dosyasından bir kareyi alarak ve bu kareyi görselleştirerek işlemleri gerçekleştiren display_image_from_video adında\n    #bir fonksiyon tanımlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:34.057331Z","iopub.execute_input":"2024-03-31T01:17:34.057662Z","iopub.status.idle":"2024-03-31T01:17:34.064054Z","shell.execute_reply.started":"2024-03-31T01:17:34.057604Z","shell.execute_reply":"2024-03-31T01:17:34.0631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video_file))\n#Bu döngü, 'fake_train_sample_video' listesindeki her bir video dosyası için display_image_from_video fonksiyonunu\n#çağırarak her bir video dosyasından bir kareyi görselleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:36.39929Z","iopub.execute_input":"2024-03-31T01:17:36.399613Z","iopub.status.idle":"2024-03-31T01:17:38.142882Z","shell.execute_reply.started":"2024-03-31T01:17:36.399573Z","shell.execute_reply":"2024-03-31T01:17:38.142118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='REAL'].sample(3).index) #viewing the real videos\nreal_train_sample_video\n# 'REAL' etiketine sahip rastgele seçilen 3 örneğin adlarını içeren bir liste olan 'real_train_sample_video'yu oluşturur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:44.965403Z","iopub.execute_input":"2024-03-31T01:17:44.965736Z","iopub.status.idle":"2024-03-31T01:17:44.97422Z","shell.execute_reply.started":"2024-03-31T01:17:44.965686Z","shell.execute_reply":"2024-03-31T01:17:44.973273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video in real_train_sample_video:\n  display_image_from_video(os.path.join(DATA_FOLDER,TRAIN_SAMPLE_FOLDER,video))\n#real_train_sample_video' listesindeki her bir video dosyası için 'display_image_from_video'\n#fonksiyonunu çağırarak her bir video dosyasından bir kareyi görselleştir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:17:46.814503Z","iopub.execute_input":"2024-03-31T01:17:46.814841Z","iopub.status.idle":"2024-03-31T01:17:48.664352Z","shell.execute_reply.started":"2024-03-31T01:17:46.814798Z","shell.execute_reply":"2024-03-31T01:17:48.663532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    #we only show images extracted from first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n      video_path = os.path.join(DATA_FOLDER, video_folder, video_file)\n      capture_img = cv.VideoCapture(video_path)\n      ret, frame = capture_img.read()\n      frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n      ax[i//3, i%3].imshow(frame)\n      ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n      ax[i//3, i%3].axis('on')\n    #Bu kod parçacığı, bir video dosyası listesinden kareleri görselleştirmek için display_image_from_video_list adında bir fonksiyon tanımlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:27.64077Z","iopub.execute_input":"2024-03-31T01:18:27.641135Z","iopub.status.idle":"2024-03-31T01:18:27.64989Z","shell.execute_reply.started":"2024-03-31T01:18:27.641059Z","shell.execute_reply":"2024-03-31T01:18:27.649118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='meawmsgiti.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)\n#Bu kod parçacığı, 'original' sütunu 'meawmsgiti.mp4' olan örneklerin indekslerini içeren bir liste olan\n#'same_original_fake_train_sample_video'yu oluşturur ve bu örneklerin karelerini görselleştirmek için display_image_from_video_list fonksiyonunu çağırır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:29.458533Z","iopub.execute_input":"2024-03-31T01:18:29.458869Z","iopub.status.idle":"2024-03-31T01:18:32.164504Z","shell.execute_reply.started":"2024-03-31T01:18:29.458809Z","shell.execute_reply":"2024-03-31T01:18:32.163772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvf /kaggle/input/ffmpeg-static-build/ffmpeg-git-amd64-static.tar.xz","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:36.122913Z","iopub.execute_input":"2024-03-31T01:18:36.123206Z","iopub.status.idle":"2024-03-31T01:18:40.391438Z","shell.execute_reply.started":"2024-03-31T01:18:36.123162Z","shell.execute_reply":"2024-03-31T01:18:40.390531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob, shutil\nimport timeit, os, gc\nimport subprocess as sp\nfrom tqdm import tqdm\nfrom collections import defaultdict\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nimport json\nfrom IPython.display import HTML\nfrom base64 import b64encode\nimport cv2\n%matplotlib inline\n#gerekli olan kütüphane ve modelleri projemize alıyoruz.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:43.66214Z","iopub.execute_input":"2024-03-31T01:18:43.662644Z","iopub.status.idle":"2024-03-31T01:18:43.673094Z","shell.execute_reply.started":"2024-03-31T01:18:43.66252Z","shell.execute_reply":"2024-03-31T01:18:43.672346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOME = \"./\"\nFFMPEG = \"/kaggle/working/ffmpeg-git-20191209-amd64-static\"\nFFMPEG_PATH = FFMPEG\nDATA_FOLDER = \"/kaggle/input/deepfake-detection-challenge\"\nTMP_FOLDER = HOME\nDATA_FOLDER_TRAIN = DATA_FOLDER\nVIDEOS_FOLDER_TRAIN = DATA_FOLDER_TRAIN + \"/train_sample_videos\"\nIMAGES_FOLDER_TRAIN = TMP_FOLDER + \"/images\"\nAUDIOS_FOLDER_TRAIN = TMP_FOLDER + \"/audios\"\nEXTRACT_META = True # False\nEXTRACT_CONTENT = True # False\nEXTRACT_FACES = True # False\nFRAME_RATE = 0.5 # Frame per\nprint(FFMPEG)\n#Bu kod parçacığı, bir dizi değişken tanımlar ve belirli değerlere atar. ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:49.959516Z","iopub.execute_input":"2024-03-31T01:18:49.959842Z","iopub.status.idle":"2024-03-31T01:18:49.966176Z","shell.execute_reply.started":"2024-03-31T01:18:49.959782Z","shell.execute_reply":"2024-03-31T01:18:49.965416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_command(*popenargs, **kwargs):\n    closeNULL = 0\n    try:\n        from subprocess import DEVNULL\n        closeNULL = 0\n    except ImportError:\n        import os\n        DEVNULL = open(os.devnull, 'wb')\n        closeNULL = 1\n\n    process = sp.Popen(stdout=sp.PIPE, stderr=DEVNULL, *popenargs, **kwargs)\n    output, unused_err = process.communicate()\n    retcode = process.poll()\n\n    if closeNULL:\n        DEVNULL.close()\n\n    if retcode:\n        cmd = kwargs.get(\"args\")\n        if cmd is None:\n            cmd = popenargs[0]\n        error = sp.CalledProcessError(retcode, cmd)\n        error.output = output\n        raise error\n    return output\n\ndef ffprobe(filename, options = [\"-show_error\", \"-show_format\", \"-show_streams\", \"-show_programs\", \"-show_chapters\", \"-show_private_data\"]):\n    ret = {}\n    command = [FFMPEG_PATH + \"/ffprobe\", \"-v\", \"error\", *options, \"-print_format\", \"json\", filename]\n    ret = run_command(command)\n    if ret:\n        ret = json.loads(ret)\n    return ret\n\n# ffmpeg -i input.mov -r 0.25 output_%04d.png\ndef ffextract_frames(filename, output_folder, rate = 0.25):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-r\", str(rate), \"-y\", output_folder + \"/output_%04d.png\"]\n    ret = run_command(command)\n    return ret\n\n# ffmpeg -i input-video.mp4 output-audio.mp3\ndef ffextract_audio(filename, output_path):\n    command = [FFMPEG_PATH + \"/ffmpeg\", \"-i\", filename, \"-vn\", \"-ac\", \"1\", \"-acodec\", \"copy\", \"-y\", output_path]\n    ret = run_command(command)\n    return ret\n#Bu işlevler, ffmpeg'i kullanarak medya dosyaları üzerinde çeşitli işlemler gerçekleştirmek için kullanılabilir.\n#Örneğin, bir video dosyasından karelerin çıkarılması veya bir video dosyasından sesin çıkarılması gibi işlemler bu işlevler aracılığıyla yapılabilir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:52.974333Z","iopub.execute_input":"2024-03-31T01:18:52.97464Z","iopub.status.idle":"2024-03-31T01:18:52.990236Z","shell.execute_reply.started":"2024-03-31T01:18:52.974596Z","shell.execute_reply":"2024-03-31T01:18:52.989367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTRACT_META == True:\n    results = []\n    subfolder = VIDEOS_FOLDER_TRAIN\n    filepaths = glob.glob(subfolder + \"/*.mp4\")\n    for filepath in tqdm(filepaths):\n        js = ffprobe(filepath)\n#        print(js)\n        if js:\n            results.append(\n                (js.get(\"format\", {}).get(\"filename\")[len(subfolder) + 1:],\n                js.get(\"format\", {}).get(\"format_long_name\"),\n                # Video \n                js.get(\"streams\", [{}, {}])[0].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"height\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"width\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"start_time\"),\n                js.get(\"streams\", [{}, {}])[0].get(\"avg_frame_rate\"),\n                 # Audio\n                js.get(\"streams\", [{}, {}])[1].get(\"codec_name\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"channels\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"sample_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"nb_frames\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"bit_rate\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"duration\"),\n                js.get(\"streams\", [{}, {}])[1].get(\"start_time\")),\n            )\n\n    meta_pd = pd.DataFrame(results, columns=[\"filename\", \"format\", \"video_codec_name\", \"video_height\", \"video_width\",\n                                            \"video_nb_frames\", \"video_bit_rate\", \"video_duration\", \"video_start_time\",\"video_fps\",\n                                            \"audio_codec_name\", \"audio_channels\", \"audio_sample_rate\", \"audio_nb_frames\",\n                                            \"audio_bit_rate\", \"audio_duration\", \"audio_start_time\"])\n    meta_pd[\"video_fps\"] = meta_pd[\"video_fps\"].apply(lambda x: float(x.split(\"/\")[0])/float(x.split(\"/\")[1]) if len(x.split(\"/\")) == 2 else None)\n    meta_pd[\"video_duration\"] = meta_pd[\"video_duration\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"video_start_time\"] = meta_pd[\"video_start_time\"].astype(np.float32)\n    meta_pd[\"video_nb_frames\"] = meta_pd[\"video_nb_frames\"].astype(np.float32)\n    meta_pd[\"video_bit_rate\"] = meta_pd[\"video_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_sample_rate\"] = meta_pd[\"audio_sample_rate\"].astype(np.float32)\n    meta_pd[\"audio_nb_frames\"] = meta_pd[\"audio_nb_frames\"].astype(np.float32)\n    meta_pd[\"audio_bit_rate\"] = meta_pd[\"audio_bit_rate\"].astype(np.float32)\n    meta_pd[\"audio_duration\"] = meta_pd[\"audio_duration\"].astype(np.float32)\n    meta_pd[\"audio_start_time\"] = meta_pd[\"audio_start_time\"].astype(np.float32)\n    meta_pd.to_pickle(HOME + \"videos_meta.pkl\")\nelse:\n    meta_pd = pd.read_pickle(HOME + \"videos_meta.pkl\")\nmeta_pd.head()\n#Bu kod parçacığı, EXTRACT_META değişkeni True ise video dosyalarından meta verilerin çıkarılması için işlem yapar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:18:56.553714Z","iopub.execute_input":"2024-03-31T01:18:56.554057Z","iopub.status.idle":"2024-03-31T01:19:12.809956Z","shell.execute_reply.started":"2024-03-31T01:18:56.553996Z","shell.execute_reply":"2024-03-31T01:19:12.808955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,6, figsize=(22, 3))\nd = sns.distplot(meta_pd[\"video_fps\"], ax=ax[0])\nd = sns.distplot(meta_pd[\"video_duration\"], ax=ax[1])\nd = sns.distplot(meta_pd[\"video_width\"], ax=ax[2])\nd = sns.distplot(meta_pd[\"video_height\"], ax=ax[3])\nd = sns.distplot(meta_pd[\"video_nb_frames\"], ax=ax[4])\nd = sns.distplot(meta_pd[\"video_bit_rate\"], ax=ax[5])\n#Bu kod parçacığı, altı farklı özelliğin (video_fps, video_duration, video_width, video_height,\n#video_nb_frames, video_bit_rate) dağılımını görselleştirmek için altı ayrı alt grafik oluşturur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:19:19.880163Z","iopub.execute_input":"2024-03-31T01:19:19.880473Z","iopub.status.idle":"2024-03-31T01:19:21.681598Z","shell.execute_reply.started":"2024-03-31T01:19:19.880427Z","shell.execute_reply":"2024-03-31T01:19:21.680499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()\n#Bu kod parçacığı, VIDEOS_FOLDER_TRAIN içindeki bir JSON dosyasından veri yükler ve ardından bu veriyi bir DataFrame'e dönüştürür.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:19:29.544572Z","iopub.execute_input":"2024-03-31T01:19:29.544879Z","iopub.status.idle":"2024-03-31T01:19:29.71543Z","shell.execute_reply.started":"2024-03-31T01:19:29.54483Z","shell.execute_reply":"2024-03-31T01:19:29.71472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_json(VIDEOS_FOLDER_TRAIN + \"/metadata.json\").T.reset_index().rename(columns={\"index\": \"filename\"})\ntrain_pd.head()\n#Bu kod parçacığı, belirtilen JSON dosyasından veriyi yükler ve bu veriyi bir DataFrame'e dönüştürür.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:19:31.668402Z","iopub.execute_input":"2024-03-31T01:19:31.668694Z","iopub.status.idle":"2024-03-31T01:19:31.838552Z","shell.execute_reply.started":"2024-03-31T01:19:31.668652Z","shell.execute_reply":"2024-03-31T01:19:31.83787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.merge(train_pd, meta_pd[[\"filename\", \"video_height\", \"video_width\", \"video_nb_frames\", \"video_bit_rate\", \"audio_nb_frames\"]], on=\"filename\", how=\"left\")\ntrain_pd[\"count\"] = train_pd.groupby([\"original\"])[\"original\"].transform('count')\n# train_pd.to_pickle(HOME + \"train_meta.pkl\")\ntrain_pd.head()\n#Bu kod parçacığı, train_pd DataFrame'ine meta_pd DataFrame'inden bazı sütunları ekler ve birleştirir.\n#Ardından, original sütununa göre gruplar oluşturarak her bir grubun eleman sayısını hesaplar ve yeni bir count sütunu oluşturur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:19:33.808626Z","iopub.execute_input":"2024-03-31T01:19:33.80892Z","iopub.status.idle":"2024-03-31T01:19:33.8416Z","shell.execute_reply.started":"2024-03-31T01:19:33.808877Z","shell.execute_reply":"2024-03-31T01:19:33.840937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUDIO_FORMAT = \"aac\" # \"wav\"\nvideos_folder = VIDEOS_FOLDER_TRAIN\nimages_folder_path = IMAGES_FOLDER_TRAIN\naudios_folder_path = AUDIOS_FOLDER_TRAIN\nif EXTRACT_CONTENT == True:\n    # 1h20min for chunk#0 (11GB)\n    # Extract some images + audio track\n    for idx, row in tqdm(train_pd.iterrows(), total=meta_pd.shape[0]):\n        try:\n            video_path = videos_folder + \"/\" + row[\"filename\"]\n            images_path = images_folder_path + \"/\" + row[\"filename\"][:-4]\n            audio_path = audios_folder_path + \"/\" + row[\"filename\"][:-4]\n            # Extract images\n            if not os.path.exists(images_path): os.makedirs(images_path)\n            ret = ffextract_frames(video_path, images_path, rate = FRAME_RATE)\n            # Extract audio\n            if not os.path.exists(audio_path): os.makedirs(audio_path)\n            # ret = ffextract_audio(video_path, audio_path + \"/audio.\" + AUDIO_FORMAT)\n        except:\n            print(\"Cannot extract frames/audio for:\" + row[\"filename\"])\n#Bu kod parçacığı, EXTRACT_CONTENT değişkeni True olduğunda, eğitim veri setindeki her bir video dosyasından görüntüler\n#ve seslerin çıkarılması işlemini gerçekleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:19:35.910337Z","iopub.execute_input":"2024-03-31T01:19:35.910673Z","iopub.status.idle":"2024-03-31T01:31:22.591347Z","shell.execute_reply.started":"2024-03-31T01:19:35.910618Z","shell.execute_reply":"2024-03-31T01:31:22.590372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd.tail()","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:32:59.636478Z","iopub.execute_input":"2024-03-31T01:32:59.636826Z","iopub.status.idle":"2024-03-31T01:32:59.656299Z","shell.execute_reply.started":"2024-03-31T01:32:59.636775Z","shell.execute_reply":"2024-03-31T01:32:59.655316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 21 # 27 # 21 # 19 # 12 # 6\nfake = train_pd[\"filename\"][idx]\nreal = train_pd[\"original\"][idx]\nvid_width = train_pd[\"video_width\"][idx]\nvid_real = open(VIDEOS_FOLDER_TRAIN + \"/\" + real, 'rb').read()\ndata_url_real = \"data:video/mp4;base64,\" + b64encode(vid_real).decode()\nvid_fake = open(VIDEOS_FOLDER_TRAIN + \"/\" + fake, 'rb').read()\ndata_url_fake = \"data:video/mp4;base64,\" + b64encode(vid_fake).decode()\nHTML(\"\"\"\n<div style='width: 100%%; display: table;'>\n    <div style='display: table-row'>\n        <div style='width: %dpx; display: table-cell;'><b>Real</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n        <div style='display: table-cell;'><b>Fake</b>: %s<br/><video width=%d controls><source src=\"%s\" type=\"video/mp4\"></video></div>\n    </div>\n</div>\n\"\"\" % ( int(vid_width/3.2) + 10, \n       real, int(vid_width/3.2), data_url_real, \n       fake, int(vid_width/3.2), data_url_fake))\n#Bu kod parçacığı, belirli bir indeksteki eğitim veri setinden bir real ve bir fake videoyu HTML üzerinde görsel olarak gösterir.\n#İlgili real ve fake videoların adı,genişliği ve kodlanmış verisi kullanılarak HTML içeriği oluşturulur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:33:02.086879Z","iopub.execute_input":"2024-03-31T01:33:02.087207Z","iopub.status.idle":"2024-03-31T01:33:02.552382Z","shell.execute_reply.started":"2024-03-31T01:33:02.087157Z","shell.execute_reply":"2024-03-31T01:33:02.551172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + \"haarcascade_frontalface_default.xml\")\n\ndef detect_face_cv2(img):\n    # Move to grayscale\n    gray_img = cv2.cvtColor(img.copy(), cv2.COLOR_RGB2GRAY)\n    face_locations = []\n    face_rects = face_cascade.detectMultiScale(gray_img, scaleFactor=1.3, minNeighbors=5)     \n    for (x,y,w,h) in face_rects: \n        face_location = (x,y,w,h)\n        face_locations.append((face_location, 1.0))\n    return face_locations\n#Bu kod parçacığı, OpenCV kütüphanesi kullanılarak bir görüntüdeki yüzleri algılamak için bir fonksiyon tanımlar.\n#detect_face_cv2 fonksiyonu, giriş olarak bir görüntü alır ve bu görüntüdeki yüzleri belirler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:33:48.447908Z","iopub.execute_input":"2024-03-31T01:33:48.448258Z","iopub.status.idle":"2024-03-31T01:33:48.480593Z","shell.execute_reply.started":"2024-03-31T01:33:48.448197Z","shell.execute_reply":"2024-03-31T01:33:48.479765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mtcnn","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:33:50.862623Z","iopub.execute_input":"2024-03-31T01:33:50.862958Z","iopub.status.idle":"2024-03-31T01:33:57.399474Z","shell.execute_reply.started":"2024-03-31T01:33:50.862892Z","shell.execute_reply":"2024-03-31T01:33:57.398609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mtcnn import MTCNN\ndetector = MTCNN()\n\ndef detect_face_mtcnn(img):\n    face_locations = []\n    items = detector.detect_faces(img)\n    for face in items:\n        face_location = tuple(face.get('box'))\n        face_confidence = float(face.get('confidence'))\n        face_locations.append((face_location, face_confidence))\n    return face_locations\n#Bu kod parçacığı, MTCNN (Multi-Task Cascaded Convolutional Networks) modeli kullanarak bir görüntüdeki yüzleri tespit etmek için bir fonksiyon tanımlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:34:00.562264Z","iopub.execute_input":"2024-03-31T01:34:00.562601Z","iopub.status.idle":"2024-03-31T01:34:11.485485Z","shell.execute_reply.started":"2024-03-31T01:34:00.562551Z","shell.execute_reply":"2024-03-31T01:34:11.48482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_faces(files, source, detector=detect_face_cv2):\n    results = []\n    # for idx, file in tqdm(enumerate(files), total=len(files)):\n    for idx, file in enumerate(files):\n        try:\n            img = cv2.cvtColor(cv2.imread(file, cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)\n            face_locations = detector(img)\n            results.append((source, file[file.find(\"output_\"):], face_locations, len(face_locations)))\n        except:\n            print(\"Cannot extract faces for image: %s\" % file)\n    return results\n#Bu kod parçacığı, verilen bir dosya listesinden yüzlerin çıkarılması işlemini gerçekleştiren bir fonksiyon tanımlar.\n#Fonksiyon, her bir dosya için yüz tespiti yapar ve tespit edilen yüzlerin konumlarını ve sayısını kaydeder.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:34:15.491894Z","iopub.execute_input":"2024-03-31T01:34:15.492224Z","iopub.status.idle":"2024-03-31T01:34:15.499493Z","shell.execute_reply.started":"2024-03-31T01:34:15.492172Z","shell.execute_reply":"2024-03-31T01:34:15.498471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = fake\ndump_folder = images_folder_path + \"/\" + file[:-4]\nfiles = glob.glob(dump_folder + \"/*\")\nDETECTORS = {\n    \"cv2\": detect_face_cv2,\n    \"mtcnn\": detect_face_mtcnn\n}\nfaces_pd = None\nfor key, value in DETECTORS.items():\n    tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n    if faces_pd is None:\n        faces_pd = tmp_pd\n    else:\n        faces_pd = pd.merge(faces_pd, tmp_pd, on=[\"filename\", \"image\"], how=\"left\")\nfaces_pd.head(12)\n#Bu kod parçacığı, belirli bir dosyadan (fake) çıkarılan yüzlerin özelliklerini içeren bir veri çerçevesi oluşturur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:34:18.790933Z","iopub.execute_input":"2024-03-31T01:34:18.791257Z","iopub.status.idle":"2024-03-31T01:34:28.417452Z","shell.execute_reply.started":"2024-03-31T01:34:18.791208Z","shell.execute_reply":"2024-03-31T01:34:28.416559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_faces_boxes(df, max_cols = 2, max_rows = 6, fsize=(24, 5), max_items=12):    \n    idx = 0    \n    for item_idx, item in df.iterrows():\n        img = cv2.cvtColor(cv2.imread(IMAGES_FOLDER_TRAIN + \"/\" + item[\"filename\"][:-4] +\"/\" + item[\"image\"], cv2.IMREAD_UNCHANGED), cv2.COLOR_BGR2RGB)    \n        face_img = img #.copy()\n        # grid subplots\n        row = idx // max_cols\n        col = idx % max_cols\n        if col == 0: fig = plt.figure(figsize=fsize)\n        ax = fig.add_subplot(1, max_cols, col + 1)\n        ax.axis(\"off\")\n        # display image with boxes\n        cols = [c for c in df.columns if \"boxes\" in c]\n        for i, c in enumerate(cols, 0):\n            face_locations = item[c]\n            face_confidence = item[c]            \n            if len(face_locations) > 0:\n                for face_location in face_locations:        \n                    ((x,y,w,h), confidence) = face_location\n                    # face_img = face_img[y:y+h, x:x+w]\n                    cv2.rectangle(face_img, (x, y), (x+w, y+h), (255,i*255,0), 8)\n                    cv2.putText(face_img, '%.1f' % (confidence*100.0), (x+w, y+h), cv2.FONT_HERSHEY_SIMPLEX, 2.0, (255,i*255,0), 9, cv2.LINE_AA)\n                ax.imshow(face_img)\n            else:\n                ax.imshow(img)\n            ax.set_title(\"%s %s / %s - Faces: %d %s %s\" % (item[\"label\"] if \"label\" in df.columns else \"\", \n                                                           item[\"filename\"], item[\"image\"],\n                                                           item[\"faces_mtcnn\"] if \"faces_mtcnn\" in df.columns else len(face_locations),\n                                                           item[\"faces_mtcnn_median\"] if \"faces_mtcnn_median\" in df.columns else \"\",\n                                                           item[\"faces\"] if \"faces\" in df.columns else \"\"))\n        if (col == max_cols -1): plt.show()\n        idx = idx + 1\n        if (max_items > 0 and idx >=max_items): break\n#Bu kod parçacığı, bir veri çerçevesindeki yüz tespiti sonuçlarını görselleştirmek için bir fonksiyon tanımlar.\n#Bu fonksiyon, her bir görüntü için tespit edilen yüzleri ve bu yüzlerin çerçeveye alındığı kutuları görselleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:34:32.489409Z","iopub.execute_input":"2024-03-31T01:34:32.489743Z","iopub.status.idle":"2024-03-31T01:34:32.517013Z","shell.execute_reply.started":"2024-03-31T01:34:32.489683Z","shell.execute_reply":"2024-03-31T01:34:32.516014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(faces_pd)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:34:35.891143Z","iopub.execute_input":"2024-03-31T01:34:35.89148Z","iopub.status.idle":"2024-03-31T01:34:38.58457Z","shell.execute_reply.started":"2024-03-31T01:34:35.891424Z","shell.execute_reply":"2024-03-31T01:34:38.583787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_detector_on_video(videos_filename, verbose=False):\n    if verbose == True: \n        print(\"Starting with batch of %d videos\" % len(videos_filename))\n    tmp_faces_pd = None\n    for file in videos_filename:\n        # Find out dump folder with images\n        dump_folder = images_folder_path + \"/\" + file[:-4]\n        # List files\n        files = glob.glob(dump_folder + \"/*\")\n        DETECTORS = {\n            \"mtcnn\": detect_face_mtcnn\n        }\n        for key, value in DETECTORS.items():\n            tmp_pd = pd.DataFrame(extract_faces(files, file, detector=value), columns=[\"filename\", \"image\", \"boxes_\" + key , \"faces_\" + key])\n            if tmp_faces_pd is None:\n                tmp_faces_pd = tmp_pd\n            else:\n                tmp_faces_pd = pd.concat([tmp_faces_pd, tmp_pd], axis=0)\n    return tmp_faces_pd\n#Bu fonksiyon, bir video dosyasından yüzlerin tespit edilmesini sağlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T01:34:52.894603Z","iopub.execute_input":"2024-03-31T01:34:52.894893Z","iopub.status.idle":"2024-03-31T01:34:52.903601Z","shell.execute_reply.started":"2024-03-31T01:34:52.894855Z","shell.execute_reply":"2024-03-31T01:34:52.902821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Z?Ğ_0   import multiprocessing\ncpus = multiprocessing.cpu_count()\nif EXTRACT_FACES == True:\n    resultfutures = []\n    results = []\n    tasks = np.array_split(train_pd[\"filename\"].unique(), 20)\n    print(\"Tasks: %d\" % len(tasks))\n    with ThreadPoolExecutor(max_workers=cpus) as executor:\n        resultfutures = tqdm(executor.map(run_detector_on_video, tasks), total=len(tasks))\n    results = [x for x in resultfutures]\n    executor.shutdown()\n    # Gather results\n    all_faces_pd = None\n    for result in results:\n        if all_faces_pd is None:\n            all_faces_pd = result\n        else:\n            all_faces_pd = pd.concat([all_faces_pd, result], axis=0)\n    all_faces_pd = all_faces_pd.reset_index(drop=True)\n    all_faces_pd.to_pickle(HOME + \"faces.pkl\")\nelse:\n    all_faces_pd = pd.read_pickle(HOME + \"faces.pkl\")\nprint(all_faces_pd.shape)\n#\nBu kod parçası, yüzlerin tespit edilmesini çoklu işlemle paralel hale getirir ve sonuçları bir veri çerçevesinde toplar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:05:34.178453Z","iopub.execute_input":"2024-03-31T02:05:34.178774Z","iopub.status.idle":"2024-03-31T02:16:59.082511Z","shell.execute_reply.started":"2024-03-31T02:05:34.178729Z","shell.execute_reply":"2024-03-31T02:16:59.074245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_faces_pd[\"faces_mtcnn_avg\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmean)\nall_faces_pd[\"faces_mtcnn_median\"] = all_faces_pd.groupby(\"filename\")[\"faces_mtcnn\"].transform(np.nanmedian)\nall_faces_pd.head()\n#Bu kod parçası, her video dosyası için MTCNN tarafından tespit edilen yüzlerin ortalama ve medyan sayısını hesaplar ve bunları yeni sütunlar olarak ekler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:17:05.402177Z","iopub.execute_input":"2024-03-31T02:17:05.402693Z","iopub.status.idle":"2024-03-31T02:17:05.476014Z","shell.execute_reply.started":"2024-03-31T02:17:05.402611Z","shell.execute_reply":"2024-03-31T02:17:05.474497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_avg\"], kde=True, ax=ax[0])\nd = sns.distplot(all_faces_pd[\"faces_mtcnn_median\"], kde=False, ax=ax[1])\n#Bu kod parçası, MTCNN tarafından tespit edilen yüzlerin ortalama ve medyan sayısının dağılımını görselleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:17:08.588647Z","iopub.execute_input":"2024-03-31T02:17:08.590191Z","iopub.status.idle":"2024-03-31T02:17:10.822644Z","shell.execute_reply.started":"2024-03-31T02:17:08.590128Z","shell.execute_reply":"2024-03-31T02:17:10.81274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_faces_boxes(all_faces_pd[all_faces_pd[\"faces_mtcnn\"] == 3], max_items=24)\n#Bu kod, MTCNN tarafından tespit edilen ve yüz sayısı 3 olan görüntülerin kutularını ve yüzlerini görselleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:17:13.537537Z","iopub.execute_input":"2024-03-31T02:17:13.538671Z","iopub.status.idle":"2024-03-31T02:17:26.177761Z","shell.execute_reply.started":"2024-03-31T02:17:13.53842Z","shell.execute_reply":"2024-03-31T02:17:26.168567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_faces_pd = pd.merge(all_faces_pd, train_pd, on=\"filename\", how=\"left\")\nclean_faces_pd.head()\n#Bu kod, tüm yüz verilerini ve eğitim verilerini birleştirerek temizlenmiş bir yüz veri çerçevesi oluşturur.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:17:51.828342Z","iopub.execute_input":"2024-03-31T02:17:51.828905Z","iopub.status.idle":"2024-03-31T02:17:51.893357Z","shell.execute_reply.started":"2024-03-31T02:17:51.828656Z","shell.execute_reply":"2024-03-31T02:17:51.88881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def faces_max_item(boxes, idx1, idx2):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[idx1][idx2])[idx1][idx2]\n    return ret\n\ndef faces_max_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = max(boxes, key=lambda item: item[1])[1]\n    return ret\n\ndef faces_min_confidence(boxes):\n    ret = 0\n    if len(boxes) > 0:\n        ret = min(boxes, key=lambda item: item[1])[1]\n    return ret\n\nclean_faces_pd[\"faces_max_width\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 2)) \nclean_faces_pd[\"faces_max_height\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_item(x, 0, 3))\nclean_faces_pd[\"faces_max_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_max_confidence(x))\nclean_faces_pd[\"faces_min_conf\"] = clean_faces_pd[\"boxes_mtcnn\"].apply(lambda x: faces_min_confidence(x))\n#Bu kod, her bir yüzün maksimum genişliği, maksimum yüksekliği ve maksimum güven değeri gibi özelliklerini hesaplar ve temizlenmiş yüz veri çerçevesine ekler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:18:00.183403Z","iopub.execute_input":"2024-03-31T02:18:00.184744Z","iopub.status.idle":"2024-03-31T02:18:00.240774Z","shell.execute_reply.started":"2024-03-31T02:18:00.184463Z","shell.execute_reply":"2024-03-31T02:18:00.235228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Faces stats:\")\nprint(clean_faces_pd[[\"faces_max_width\", \"faces_max_height\", \"faces_min_conf\", \"faces_max_conf\"]].describe(percentiles=[0.01,0.05, 0.1,0.25,0.5,0.75,0.9,0.95,0.99]))\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_max_width\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_height\"], kde=True, ax=ax[1])\nplt.show()\nfig, ax = plt.subplots(1, 2, figsize=(22, 3))\nd = sns.distplot(clean_faces_pd[\"faces_min_conf\"], kde=True, ax=ax[0])\nd = sns.distplot(clean_faces_pd[\"faces_max_conf\"], kde=True, ax=ax[1])\nfig, ax = plt.subplots(figsize=(22, 3))\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_width\", y=\"faces_max_conf\", c=\"red\", ax=ax, label=\"faces_max_width\", alpha=0.5)\nd = clean_faces_pd.plot(kind=\"scatter\", x=\"faces_max_height\", y=\"faces_max_conf\", c=\"blue\", ax=d,  label=\"faces_max_height\", alpha=0.5)\nd = plt.legend(loc=\"upper right\")\n#Bu kod, temizlenmiş yüz veri çerçevesinin bazı istatistiklerini hesaplar ve bu verileri görselleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:18:03.326283Z","iopub.execute_input":"2024-03-31T02:18:03.326624Z","iopub.status.idle":"2024-03-31T02:18:10.083301Z","shell.execute_reply.started":"2024-03-31T02:18:03.326573Z","shell.execute_reply":"2024-03-31T02:18:10.079043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install imutils","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:18:15.539461Z","iopub.execute_input":"2024-03-31T02:18:15.540684Z","iopub.status.idle":"2024-03-31T02:18:26.480802Z","shell.execute_reply.started":"2024-03-31T02:18:15.540262Z","shell.execute_reply":"2024-03-31T02:18:26.475884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport shutil\nimport cv2\nimport pandas as pd\nimport matplotlib\nmatplotlib.use(\"Agg\")\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pickle\nfrom imutils import paths\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import VGG16\nfrom keras.layers.core import Dropout\nfrom keras.layers.core import Flatten\nfrom keras.layers.core import Dense\nfrom keras.layers import Input\nfrom keras.models import Model\nfrom keras.optimizers import SGD\nfrom sklearn.metrics import classification_report\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:18:35.707957Z","iopub.execute_input":"2024-03-31T02:18:35.708605Z","iopub.status.idle":"2024-03-31T02:18:37.469528Z","shell.execute_reply.started":"2024-03-31T02:18:35.708537Z","shell.execute_reply":"2024-03-31T02:18:37.466948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/working/finetuningkeras/dataset\"\n\n# define the names of the training, testing, and validation\n# directories\nTRAIN = \"training\"\nTEST = \"evaluation\"\nVAL = \"validation\"\n\nREAL = 'REAL'\nFAKE = 'FAKE'\n\n# initialize the list of class label names\nCLASSES = [\"FAKE\", \"REAL\"]\n\n\n# set the batch size when fine-tuning\nBATCH_SIZE = 32\n\ntrainEpochs = 10\nepochsFineTune = 10\nmaxVids = 5\n\n# set the path to the serialized model after training\nMODEL_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"Deepfake.model\"])\n\n# define the path to the output training history plots\nUNFROZEN_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"unfrozen.png\"])\nWARMUP_PLOT_PATH = os.path.sep.join([\"/kaggle/working/finetuningkeras\",\"output\", \"warmup.png\"])\n\nfile = '/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json'\nimg_path = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\ndata_path = '/kaggle/working/finetuningkeras/real_fake'\ndir_fake_frames = '/kaggle/working/FAKE_frames'\ndir_real_frames = '/kaggle/working/REAL_frames'\ndir_output = '/kaggle/working/finetuningkeras/output'\n\ndir_data_path_real = os.path.join(data_path, REAL)\ndir_data_path_fake = os.path.join(data_path, FAKE)\n\ndir_train_real = os.path.join(BASE_PATH, TRAIN, REAL)\ndir_train_fake = os.path.join(BASE_PATH, TRAIN, FAKE)\ndir_valid_real = os.path.join(BASE_PATH, VAL, REAL)\ndir_valid_fake = os.path.join(BASE_PATH, VAL, FAKE)\ndir_test_real = os.path.join(BASE_PATH, TEST, REAL)\ndir_test_fake = os.path.join(BASE_PATH, TEST, FAKE)\n#Bu kodda, veri seti ve model eğitimiyle ilgili bazı önemli parametreler ve yol tanımları yapılmıştır. ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:18:58.744593Z","iopub.execute_input":"2024-03-31T02:18:58.745411Z","iopub.status.idle":"2024-03-31T02:18:58.823564Z","shell.execute_reply.started":"2024-03-31T02:18:58.745327Z","shell.execute_reply":"2024-03-31T02:18:58.821516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = '/kaggle/working/finetuningkeras/real_fake/FAKE'\noutput_dir = '/kaggle/working/FAKE_frames/'\ndef explode_frames(input_dir, output_dir, maxN):\n\n    mp4_filenames = [f for f in os.listdir(input_dir) if f.endswith('.mp4')]\n    n = 0\n    \n    for mp4fn in mp4_filenames:\n        \n        if(n < maxN):\n            n += 1 \n            mp4fp = os.path.join(input_dir, mp4fn)\n            cam = cv2.VideoCapture(mp4fp) \n            if(cam.isOpened()):\n                print('Processing file #'+ str(n) + ' (' + mp4fn + ')...')\n            else: \n                print('Problem opening file #'+ str(n) + ' (' + mp4fn + ')...')\n                continue \n            \n            nframe = 0\n            while(True): #continue until ret = False then break\n                nframe += 1\n                ret,frame = cam.read()\n                \n                if ret: \n                    # if video is still left continue creating images \n                    out_filename = os.path.splitext(mp4fn)[0]+  '_frame' + str(nframe) + '.jpg'\n                    out_filepath =  os.path.join(output_dir, out_filename)\n                    \n                    # writing the extracted images \n                    cv2.imwrite(out_filepath, frame) \n                else: \n                    break\n\n            # Release all space and windows once done\n            print(' - created ' + str(nframe-1) + ' images') # -1 bc count incremented before exit\n            cam.release() \n            cv2.destroyAllWindows()\n            \n        else: \n            break\n            \n\"\"\"\nDistribute files/images from a source directory into training, validation, and testing directories. \nsrc_dir = source/input directory\ntrain_dir, val_dir, test_dir = target training/validation/testing directory\nvalperc = fraction of dataset to use for validation (0-1)\ntestperc = fraction of dataset to use for testing (0-1)\n\"\"\"\n            \ndef trainvaltest_split(src_dir, train_dir, val_dir, test_dir, valperc = 0.15, testperc = 0.15):\n    \n    filenames = os.listdir(src_dir) #get all filenames in random order\n    np.random.shuffle(filenames)\n    \n    n = len(filenames)\n    split1 = int(n*(1 - (valperc + testperc)))\n    split2 = int(n*(1 - (testperc)))\n    \n    fn_train, fn_val, fn_test = np.split(np.array(filenames), [split1, split2])\n    \n    fn_lists = [fn_train, fn_val, fn_test]\n    targetdirs = [train_dir, val_dir, test_dir]\n    \n    print('Total images: ', n)\n    print('Training: ', len(fn_train))\n    print('Validation: ', len(fn_val))\n    print('Testing: ', len(fn_test))\n    \n    all_fp = [os.path.join(src_dir, fn) for fn in filenames]\n    \n    #move files\n    for i, fn_list in enumerate(fn_lists):\n        for fn in fn_list: \n            target_dir = targetdirs[i]\n            fp_from = os.path.join(src_dir, fn)\n            fp_to = os.path.join(target_dir, fn)\n            \n            shutil.move(fp_from, fp_to)\n\n            \n\"\"\"\nConstruct a plot that plots and saves the training history\n\"\"\"           \ndef plot_training(H, N, plotPath):\n\tplt.style.use(\"ggplot\")\n\tplt.figure()\n\tplt.plot(np.arange(0, N), H.history[\"loss\"], label=\"train_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"val_loss\"], label=\"val_loss\")\n\tplt.plot(np.arange(0, N), H.history[\"accuracy\"], label=\"train_acc\")\n\tplt.plot(np.arange(0, N), H.history[\"val_accuracy\"], label=\"val_acc\")\n\tplt.title(\"Training Loss and Accuracy\")\n\tplt.xlabel(\"Epoch #\")\n\tplt.ylabel(\"Loss/Accuracy\")\n\tplt.legend(loc=\"lower left\")\n\tplt.savefig(plotPath)\n#Bu kod, çerçeveleri çıkarma, veri kümelerini eğitim, doğrulama ve test setlerine bölme ve eğitim geçmişi grafiği oluşturma gibi işlevleri gerçekleştirir.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:19:05.989631Z","iopub.execute_input":"2024-03-31T02:19:05.990875Z","iopub.status.idle":"2024-03-31T02:19:06.040601Z","shell.execute_reply.started":"2024-03-31T02:19:05.990803Z","shell.execute_reply":"2024-03-31T02:19:06.032524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nos.makedirs(dir_train_real, exist_ok = True)\nos.makedirs(dir_train_fake, exist_ok = True)\nos.makedirs(dir_valid_real, exist_ok = True)\nos.makedirs(dir_valid_fake, exist_ok = True)\nos.makedirs(dir_test_real, exist_ok = True)\nos.makedirs(dir_test_fake, exist_ok = True)\n\nos.makedirs(dir_data_path_real, exist_ok = True)\nos.makedirs(dir_data_path_fake, exist_ok = True)\nos.makedirs(dir_fake_frames, exist_ok = True) \nos.makedirs(dir_real_frames, exist_ok = True) \nos.makedirs(dir_output, exist_ok = True)\n#Yukarıdaki kod, belirtilen dizinlerin hiyerarşisini oluşturur veya zaten varsa bu dizinleri geçer. ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:19:16.098807Z","iopub.execute_input":"2024-03-31T02:19:16.100472Z","iopub.status.idle":"2024-03-31T02:19:16.127439Z","shell.execute_reply.started":"2024-03-31T02:19:16.100404Z","shell.execute_reply":"2024-03-31T02:19:16.123976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_json(file)\ndf = df.T\n\n# %% [code]\nlabel = df[['label']]","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:19:18.940647Z","iopub.execute_input":"2024-03-31T02:19:18.941717Z","iopub.status.idle":"2024-03-31T02:19:19.176332Z","shell.execute_reply.started":"2024-03-31T02:19:18.941629Z","shell.execute_reply":"2024-03-31T02:19:19.171004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fn, row in label.iterrows():\n    src = os.path.join(img_path, fn)\n    dest = os.path.join(data_path, row['label'], fn)\n    shutil.copy(src, dest)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:19:22.725599Z","iopub.execute_input":"2024-03-31T02:19:22.728426Z","iopub.status.idle":"2024-03-31T02:20:08.428351Z","shell.execute_reply.started":"2024-03-31T02:19:22.726025Z","shell.execute_reply":"2024-03-31T02:20:08.423677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_fake, dir_fake_frames, maxN= maxVids)\n#Bu kod, belirtilen sahte video dosyalarını kare kare ayırarak çerçevelere dönüştürür.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:20:14.110118Z","iopub.execute_input":"2024-03-31T02:20:14.112873Z","iopub.status.idle":"2024-03-31T02:21:34.285325Z","shell.execute_reply.started":"2024-03-31T02:20:14.111569Z","shell.execute_reply":"2024-03-31T02:21:34.280866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explode_frames(dir_data_path_real, dir_real_frames, maxN= maxVids)\n#Bu kod, gerçek video dosyalarını kare kare ayırarak çerçevelere dönüştürür.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:21:43.651316Z","iopub.execute_input":"2024-03-31T02:21:43.652546Z","iopub.status.idle":"2024-03-31T02:23:04.771229Z","shell.execute_reply.started":"2024-03-31T02:21:43.652475Z","shell.execute_reply":"2024-03-31T02:23:04.763214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_fake_frames,\n                   train_dir = dir_train_fake, \n                   val_dir = dir_valid_fake, \n                   test_dir = dir_test_fake)\n#Bu kod, belirli bir kaynak dizinindeki çerçeveleri eğitim, doğrulama ve test dizinlerine dağıtarak veri kümesini böler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:09.640905Z","iopub.execute_input":"2024-03-31T02:23:09.643003Z","iopub.status.idle":"2024-03-31T02:23:10.457733Z","shell.execute_reply.started":"2024-03-31T02:23:09.642252Z","shell.execute_reply":"2024-03-31T02:23:10.451708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainvaltest_split(src_dir = dir_real_frames,\n                   train_dir = dir_train_real, \n                   val_dir = dir_valid_real, \n                   test_dir = dir_test_real)\n#Bu kod bloğu, gerçek çerçevelerin bulunduğu dizindeki çerçeveleri eğitim, doğrulama ve test dizinlerine dağıtarak gerçek\n#ve sahte çerçevelerin ayrı ayrı veri kümelerine bölünmesini sağlar. ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:13.809222Z","iopub.execute_input":"2024-03-31T02:23:13.809815Z","iopub.status.idle":"2024-03-31T02:23:14.915992Z","shell.execute_reply.started":"2024-03-31T02:23:13.809747Z","shell.execute_reply":"2024-03-31T02:23:14.911618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrainAug = ImageDataGenerator(\n\trotation_range=30,\n\tzoom_range=0.15,\n\twidth_shift_range=0.2,\n\theight_shift_range=0.2,\n\tshear_range=0.15,\n\thorizontal_flip=True,\n\tfill_mode=\"nearest\")\n#Bu kod bloğu, eğitim veri artırma işlemi için bir ImageDataGenerator nesnesi oluşturur.\n#Veri artırma, eğitim veri setinin çeşitliliğini artırarak modelin genelleme yeteneğini artırmaya yardımcı olur ve aşırı uyum riskini azaltır.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:18.128927Z","iopub.execute_input":"2024-03-31T02:23:18.12935Z","iopub.status.idle":"2024-03-31T02:23:18.142957Z","shell.execute_reply.started":"2024-03-31T02:23:18.129286Z","shell.execute_reply":"2024-03-31T02:23:18.139812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valAug = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:24.868047Z","iopub.execute_input":"2024-03-31T02:23:24.868429Z","iopub.status.idle":"2024-03-31T02:23:24.888004Z","shell.execute_reply.started":"2024-03-31T02:23:24.868368Z","shell.execute_reply":"2024-03-31T02:23:24.884677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = np.array([123.68, 116.779, 103.939], dtype=\"float32\")\ntrainAug.mean = mean\nvalAug.mean = mean\n#Bu kod bloğu, her iki veri artırma jeneratörü için de ortalama (mean) görüntü piksel değerlerini ayarlar.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:27.172998Z","iopub.execute_input":"2024-03-31T02:23:27.174194Z","iopub.status.idle":"2024-03-31T02:23:27.197493Z","shell.execute_reply.started":"2024-03-31T02:23:27.174116Z","shell.execute_reply":"2024-03-31T02:23:27.191699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainPath = os.path.join(BASE_PATH, TRAIN)\ntrainGen = trainAug.flow_from_directory(\n\ttrainPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=True,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the validation generator\nvalPath = os.path.join(BASE_PATH, VAL)\nvalGen = valAug.flow_from_directory(\n\tvalPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)\n\n# initialize the testing generator\ntestPath = os.path.join(BASE_PATH, TEST)\ntestGen = valAug.flow_from_directory(\n\ttestPath,\n\tclass_mode=\"categorical\",\n\ttarget_size=(224, 224),\n\tcolor_mode=\"rgb\",\n\tshuffle=False,\n\tbatch_size=BATCH_SIZE)\n#Bu kod bloğu, veri artırma jeneratörlerini oluşturur ve veri kümesinden görüntü verilerini yükler.","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:42.932829Z","iopub.execute_input":"2024-03-31T02:23:42.933759Z","iopub.status.idle":"2024-03-31T02:23:43.521445Z","shell.execute_reply.started":"2024-03-31T02:23:42.933601Z","shell.execute_reply":"2024-03-31T02:23:43.516553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseModel = VGG16(weights=\"imagenet\", include_top=False,\n\tinput_tensor=Input(shape=(224, 224, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:46.443315Z","iopub.execute_input":"2024-03-31T02:23:46.445252Z","iopub.status.idle":"2024-03-31T02:23:53.208638Z","shell.execute_reply.started":"2024-03-31T02:23:46.445182Z","shell.execute_reply":"2024-03-31T02:23:53.202289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headModel = baseModel.output\nheadModel = Flatten(name=\"flatten\")(headModel)\nheadModel = Dense(512, activation=\"relu\")(headModel)\nheadModel = Dropout(0.5)(headModel)\nheadModel = Dense(len(CLASSES), activation=\"softmax\")(headModel)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:23:58.398732Z","iopub.execute_input":"2024-03-31T02:23:58.400001Z","iopub.status.idle":"2024-03-31T02:23:58.586711Z","shell.execute_reply.started":"2024-03-31T02:23:58.399905Z","shell.execute_reply":"2024-03-31T02:23:58.585548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=baseModel.input, outputs=headModel)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:24:01.46143Z","iopub.execute_input":"2024-03-31T02:24:01.461868Z","iopub.status.idle":"2024-03-31T02:24:01.480711Z","shell.execute_reply.started":"2024-03-31T02:24:01.461781Z","shell.execute_reply":"2024-03-31T02:24:01.475711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers:\n\tlayer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:24:03.912866Z","iopub.execute_input":"2024-03-31T02:24:03.915305Z","iopub.status.idle":"2024-03-31T02:24:03.926515Z","shell.execute_reply.started":"2024-03-31T02:24:03.915233Z","shell.execute_reply":"2024-03-31T02:24:03.924225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:24:06.559046Z","iopub.execute_input":"2024-03-31T02:24:06.559527Z","iopub.status.idle":"2024-03-31T02:24:06.75444Z","shell.execute_reply.started":"2024-03-31T02:24:06.559455Z","shell.execute_reply":"2024-03-31T02:24:06.750045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"totalTrain = len(list(paths.list_images(trainPath)))\ntotalVal = len(list(paths.list_images(valPath)))\ntotalTest = len(list(paths.list_images(testPath)))\n\nprint(\"[INFO] training head...\")\nH = model.fit(\n    trainGen,\n    steps_per_epoch=totalTrain // BATCH_SIZE,\n    validation_data=valGen,\n    validation_steps=totalVal // BATCH_SIZE,\n    epochs=trainEpochs)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:24:11.112572Z","iopub.execute_input":"2024-03-31T02:24:11.113961Z","iopub.status.idle":"2024-03-31T02:41:21.213244Z","shell.execute_reply.started":"2024-03-31T02:24:11.113019Z","shell.execute_reply":"2024-03-31T02:41:21.21202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network head...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\n\n\nplot_training(H, trainEpochs, WARMUP_PLOT_PATH)\nplt.show()  ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:41:42.802845Z","iopub.execute_input":"2024-03-31T02:41:42.803175Z","iopub.status.idle":"2024-03-31T02:41:54.502547Z","shell.execute_reply.started":"2024-03-31T02:41:42.803129Z","shell.execute_reply":"2024-03-31T02:41:54.501443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:43:41.072477Z","iopub.execute_input":"2024-03-31T02:43:41.072794Z","iopub.status.idle":"2024-03-31T02:43:42.759783Z","shell.execute_reply.started":"2024-03-31T02:43:41.072749Z","shell.execute_reply":"2024-03-31T02:43:42.758876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns ","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:43:54.861506Z","iopub.execute_input":"2024-03-31T02:43:54.861838Z","iopub.status.idle":"2024-03-31T02:43:54.866618Z","shell.execute_reply.started":"2024-03-31T02:43:54.861775Z","shell.execute_reply":"2024-03-31T02:43:54.865545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming testGen and model are defined\ntestGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\ny_true = testGen.classes\n\n# Compute ROC curve for the positive class\nfpr, tpr, _ = roc_curve(y_true, y_pred_probs[:, 1])\nroc_auc = auc(fpr, tpr)\n\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=1, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:43:57.606612Z","iopub.execute_input":"2024-03-31T02:43:57.606947Z","iopub.status.idle":"2024-03-31T02:44:09.07121Z","shell.execute_reply.started":"2024-03-31T02:43:57.606893Z","shell.execute_reply":"2024-03-31T02:44:09.070097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testGen.reset()\ny_pred_probs = model.predict(testGen, steps=(totalTest // BATCH_SIZE) + 1)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:45:06.505405Z","iopub.execute_input":"2024-03-31T02:45:06.505984Z","iopub.status.idle":"2024-03-31T02:45:17.696563Z","shell.execute_reply.started":"2024-03-31T02:45:06.505759Z","shell.execute_reply":"2024-03-31T02:45:17.695859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:45:21.278773Z","iopub.execute_input":"2024-03-31T02:45:21.279207Z","iopub.status.idle":"2024-03-31T02:45:21.573528Z","shell.execute_reply.started":"2024-03-31T02:45:21.279139Z","shell.execute_reply":"2024-03-31T02:45:21.572431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:45:42.880152Z","iopub.execute_input":"2024-03-31T02:45:42.880604Z","iopub.status.idle":"2024-03-31T02:45:43.592052Z","shell.execute_reply.started":"2024-03-31T02:45:42.880423Z","shell.execute_reply":"2024-03-31T02:45:43.59092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainGen.reset()\nvalGen.reset()","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:45:46.326827Z","iopub.execute_input":"2024-03-31T02:45:46.327136Z","iopub.status.idle":"2024-03-31T02:45:46.331056Z","shell.execute_reply.started":"2024-03-31T02:45:46.327092Z","shell.execute_reply":"2024-03-31T02:45:46.330322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in baseModel.layers[15:]:\n\tlayer.trainable = True\n\n# loop over the layers in the model and show which ones are trainable\n# or not\nfor layer in baseModel.layers:\n\tprint(\"{}: {}\".format(layer, layer.trainable))\n\n# for the changes to the model to take affect we need to recompile\n# the model, this time using SGD with a *very* small learning rate\nprint(\"[INFO] re-compiling model...\")\nopt = SGD(lr=1e-4, momentum=0.9)\nmodel.compile(loss=\"categorical_crossentropy\", optimizer=opt,\n\tmetrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:45:49.54782Z","iopub.execute_input":"2024-03-31T02:45:49.548182Z","iopub.status.idle":"2024-03-31T02:45:49.608523Z","shell.execute_reply.started":"2024-03-31T02:45:49.548116Z","shell.execute_reply":"2024-03-31T02:45:49.607534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"H = model.fit_generator(\n\ttrainGen,\n\tsteps_per_epoch=totalTrain // BATCH_SIZE,\n\tvalidation_data=valGen,\n\tvalidation_steps=totalVal // BATCH_SIZE,\n\tepochs= epochsFineTune)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:45:54.135893Z","iopub.execute_input":"2024-03-31T02:45:54.136211Z","iopub.status.idle":"2024-03-31T02:49:08.146615Z","shell.execute_reply.started":"2024-03-31T02:45:54.136165Z","shell.execute_reply":"2024-03-31T02:49:08.144586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"[INFO] evaluating after fine-tuning network...\")\ntestGen.reset()\npredIdxs = model.predict_generator(testGen,\n\tsteps=(totalTest // BATCH_SIZE) + 1)\npredIdxs = np.argmax(predIdxs, axis=1)\nprint(classification_report(testGen.classes, predIdxs,\n\ttarget_names=testGen.class_indices.keys()))\nplot_training(H, epochsFineTune, UNFROZEN_PLOT_PATH)\n\n# serialize the model to disk\nprint(\"[INFO] serializing network...\")\nmodel.save(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:21:35.816449Z","iopub.execute_input":"2024-02-25T15:21:35.817203Z","iopub.status.idle":"2024-02-25T15:21:49.652574Z","shell.execute_reply.started":"2024-02-25T15:21:35.816887Z","shell.execute_reply":"2024-02-25T15:21:49.651815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_epoch_vs_accuracy(H, save_path=None):\n    print(\"Starting to plot...\")  # Debugging print statement\n    epochs = len(H.history['accuracy'])  # Automatically determine the number of epochs\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, epochs + 1), H.history['accuracy'], label='Train Accuracy')\n    plt.plot(range(1, epochs + 1), H.history['val_accuracy'], label='Validation Accuracy')\n    plt.title('Epoch vs Accuracy')\n    plt.ylabel('Accuracy')\n    plt.xlabel('Epoch')\n    plt.legend()\n    \n    if save_path:\n        plt.savefig(save_path)\n        \n    plt.show()\n    print(\"Plot should be displayed above.\")  # Debugging print statement\n\n# Assuming H is defined in your existing code\nplot_epoch_vs_accuracy(H)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:21.979564Z","iopub.execute_input":"2024-03-31T02:49:21.979857Z","iopub.status.idle":"2024-03-31T02:49:22.248956Z","shell.execute_reply.started":"2024-03-31T02:49:21.979815Z","shell.execute_reply":"2024-03-31T02:49:22.248092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted class labels\ny_pred_labels = np.argmax(y_pred_probs, axis=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_true, y_pred_labels)\n\n# Plot confusion matrix\nplt.figure(figsize=(5,5))\nsns.heatmap(cm, annot=True, fmt=\"d\")\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('True')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:26.585313Z","iopub.execute_input":"2024-03-31T02:49:26.585606Z","iopub.status.idle":"2024-03-31T02:49:26.861003Z","shell.execute_reply.started":"2024-03-31T02:49:26.585563Z","shell.execute_reply":"2024-03-31T02:49:26.859877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png')","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:30.059653Z","iopub.execute_input":"2024-03-31T02:49:30.059946Z","iopub.status.idle":"2024-03-31T02:49:30.335106Z","shell.execute_reply.started":"2024-03-31T02:49:30.059904Z","shell.execute_reply":"2024-03-31T02:49:30.334213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wave\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:37.605613Z","iopub.execute_input":"2024-03-31T02:49:37.605892Z","iopub.status.idle":"2024-03-31T02:49:37.609845Z","shell.execute_reply.started":"2024-03-31T02:49:37.605855Z","shell.execute_reply":"2024-03-31T02:49:37.608998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"margot_robbie_speech = wave.open('/kaggle/input/deep-voice-deepfake-voice-recognition/KAGGLE/AUDIO/REAL/margot-original.wav', 'rb')\nprint(margot_robbie_speech)\n\nsample_freq = margot_robbie_speech.getframerate()\nn_samples = margot_robbie_speech.getnframes()\nt_audio = n_samples/sample_freq\nn_channels = margot_robbie_speech.getnchannels()\n\nprint(\"The samping rate of the audio file is \" + str(sample_freq) + \"Hz, or \" + str(sample_freq/1000) + \"kHz\")\nprint(\"The audio contains a total of \" + str(n_samples) + \" frames or samples\")\nprint(\"The length of the audio file is \" + str(t_audio) + \" seconds\")\nprint(\"The audio file has \" + str(n_channels) + \" channels.\\n\")\n\n\n\nsignal_wave = margot_robbie_speech.readframes(n_samples)\nsignal_array = np.frombuffer(signal_wave, dtype=np.int16)\nprint(\"The signal contains a total of \" + str(signal_array.shape[0]) + \" samples.\")\nprint(\"If this value is greater than \" + str(n_samples) + \" it is due to there being multiple channels\")\nprint(\"E.g. - Samples * Channels = \" + str(n_samples*n_channels))\n\n# Split the channels\nl_channel = signal_array[0::2]\nr_channel = signal_array[1::2]","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:40.322835Z","iopub.execute_input":"2024-03-31T02:49:40.323168Z","iopub.status.idle":"2024-03-31T02:49:40.532575Z","shell.execute_reply.started":"2024-03-31T02:49:40.323109Z","shell.execute_reply":"2024-03-31T02:49:40.531604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(\"/kaggle/input/deep-voice-deepfake-voice-recognition/KAGGLE/DATASET-balanced.csv\")\n\nX = df.iloc[:,:-1]\ny = df.iloc[:,-1]\n\nprint(X.head(10))\nprint(y.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:44.41906Z","iopub.execute_input":"2024-03-31T02:49:44.419525Z","iopub.status.idle":"2024-03-31T02:49:44.548435Z","shell.execute_reply.started":"2024-03-31T02:49:44.419457Z","shell.execute_reply":"2024-03-31T02:49:44.547523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nlb = preprocessing.LabelBinarizer()\nlb.fit(y)\ny = lb.transform(y)\ny = y.ravel()\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:50.751389Z","iopub.execute_input":"2024-03-31T02:49:50.751719Z","iopub.status.idle":"2024-03-31T02:49:50.81287Z","shell.execute_reply.started":"2024-03-31T02:49:50.751661Z","shell.execute_reply":"2024-03-31T02:49:50.811969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier(n_estimators=80, random_state=1)\n\nfrom sklearn.model_selection import KFold\nkf = KFold(n_splits=10,  shuffle=True, random_state=1)\n\nprint(model)\nprint(\"KFold splits: \" + str(kf.get_n_splits(X)))","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:54.681091Z","iopub.execute_input":"2024-03-31T02:49:54.681413Z","iopub.status.idle":"2024-03-31T02:49:54.830938Z","shell.execute_reply.started":"2024-03-31T02:49:54.681354Z","shell.execute_reply":"2024-03-31T02:49:54.8301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\n\nimport numpy as np\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, matthews_corrcoef, roc_auc_score\n\nacc_score = []\nprec_score = []\nrec_score = []\nf1s = []\nMCCs = []\nROCareas = []\n\nstart = time.time()\nfor train_index , test_index in kf.split(X):\n    X_train , X_test = X.iloc[train_index,:],X.iloc[test_index,:]\n    y_train , y_test = y[train_index] , y[test_index]\n     \n    model.fit(X_train,y_train)\n    pred_values = model.predict(X_test)\n    acc = accuracy_score(pred_values , y_test)\n    acc_score.append(acc)\n    \n    prec = precision_score(y_test , pred_values, average=\"binary\", pos_label=1)\n    prec_score.append(prec)\n    \n    rec = recall_score(y_test , pred_values, average=\"binary\", pos_label=1)\n    rec_score.append(rec)\n    \n    f1 = f1_score(y_test , pred_values, average=\"binary\", pos_label=1)\n    f1s.append(f1)\n    \n    mcc = matthews_corrcoef(y_test , pred_values)\n    MCCs.append(mcc)   \n    \n    roc = roc_auc_score(y_test , pred_values)\n    ROCareas.append(roc)\nend = time.time()\ntimeTaken = (end - start)\nprint(\"Model trained in: \" + str( round(timeTaken, 2) ) + \" seconds.\")","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:49:57.482936Z","iopub.execute_input":"2024-03-31T02:49:57.483265Z","iopub.status.idle":"2024-03-31T02:50:34.177984Z","shell.execute_reply.started":"2024-03-31T02:49:57.483214Z","shell.execute_reply":"2024-03-31T02:50:34.177159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Mean results and (std.):\\n\")\nprint(\"Accuracy: \" + str( round(np.mean(acc_score)*100, 3) ) + \"% (\" + str( round(np.std(acc_score)*100, 3) ) + \")\\n\")\nprint(\"Precision: \" + str( round(np.mean(prec_score), 3) ) + \" (\" + str( round(np.std(prec_score), 3) ) + \")\")\nprint(\"Recall: \" + str( round(np.mean(rec_score), 3) ) + \" (\" + str( round(np.std(rec_score), 3) ) + \")\")\nprint(\"F1-Score: \" + str( round(np.mean(f1s), 3) ) + \" (\" + str( round(np.std(f1s), 3) ) + \")\")\nprint(\"MCC: \" + str( round(np.mean(MCCs), 3) ) + \" (\" + str( round(np.std(MCCs), 3) ) + \")\")\nprint(\"ROC AUC: \" + str( round(np.mean(ROCareas), 3) ) + \" (\" + str( round(np.std(ROCareas), 3) ) + \")\")","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:50:41.568213Z","iopub.execute_input":"2024-03-31T02:50:41.568632Z","iopub.status.idle":"2024-03-31T02:50:41.587945Z","shell.execute_reply.started":"2024-03-31T02:50:41.568575Z","shell.execute_reply":"2024-03-31T02:50:41.586599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport time\nimport matplotlib.pyplot as plt\n\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import KFold, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, matthews_corrcoef, roc_auc_score\n\n# Load dataset\ndf = pd.read_csv(\"/kaggle/input/deep-voice-deepfake-voice-recognition/KAGGLE/DATASET-balanced.csv\")\n\n# Data Exploration\nprint(df.head())\nprint(df.describe())\nprint(df.info())\n\n# Splitting the dataset into features and target\nX = df.iloc[:,:-1]\ny = df.iloc[:,-1]\n\n# Label Binarization\nlb = preprocessing.LabelBinarizer()\nlb.fit(y)\ny = lb.transform(y)\ny = y.ravel()\n\n# Feature scaling (Optional)\n# from sklearn.preprocessing import StandardScaler\n# scaler = StandardScaler()\n# X = scaler.fit_transform(X)\n\n# Initialize model and KFold\nmodel = RandomForestClassifier(n_estimators=80, random_state=1)\nkf = KFold(n_splits=10,  shuffle=True, random_state=1)\n\n# Metrics to keep track\nacc_score = []\nprec_score = []\nrec_score = []\nf1s = []\nMCCs = []\nROCareas = []\n\n# Model training and evaluation\nstart = time.time()\nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X.iloc[train_index,:], X.iloc[test_index,:]\n    y_train, y_test = y[train_index], y[test_index]\n    \n    model.fit(X_train, y_train)\n    pred_values = model.predict(X_test)\n    \n    acc_score.append(accuracy_score(y_test, pred_values))\n    prec_score.append(precision_score(y_test, pred_values))\n    rec_score.append(recall_score(y_test, pred_values))\n    f1s.append(f1_score(y_test, pred_values))\n    MCCs.append(matthews_corrcoef(y_test, pred_values))\n    ROCareas.append(roc_auc_score(y_test, pred_values))\n\nend = time.time()\n\n# Display Results\nprint(f\"Model trained in {round(end - start, 2)} seconds.\")\nprint(\"Mean results and (std.):\\n\")\nprint(f\"Accuracy: {round(np.mean(acc_score)*100, 3)}% ({round(np.std(acc_score)*100, 3)})\")\nprint(f\"Precision: {round(np.mean(prec_score), 3)} ({round(np.std(prec_score), 3)})\")\nprint(f\"Recall: {round(np.mean(rec_score), 3)} ({round(np.std(rec_score), 3)})\")\nprint(f\"F1-Score: {round(np.mean(f1s), 3)} ({round(np.std(f1s), 3)})\")\nprint(f\"MCC: {round(np.mean(MCCs), 3)} ({round(np.std(MCCs), 3)})\")\nprint(f\"ROC AUC: {round(np.mean(ROCareas), 3)} ({round(np.std(ROCareas), 3)})\")\n\n# Performance Visualization\nplt.figure(figsize=(12, 6))\nplt.plot(acc_score, label='Accuracy')\nplt.plot(prec_score, label='Precision')\nplt.plot(rec_score, label='Recall')\nplt.plot(f1s, label='F1 Score')\nplt.plot(MCCs, label='MCC')\nplt.plot(ROCareas, label='ROC AUC')\nplt.title('Performance metrics across folds')\nplt.xlabel('Fold')\nplt.ylabel('Metric Value')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-31T02:50:49.181466Z","iopub.execute_input":"2024-03-31T02:50:49.181798Z","iopub.status.idle":"2024-03-31T02:51:26.455785Z","shell.execute_reply.started":"2024-03-31T02:50:49.18174Z","shell.execute_reply":"2024-03-31T02:51:26.454677Z"},"trusted":true},"execution_count":null,"outputs":[]}]}