{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1>Deepfake detection\n\n   \n\n","metadata":{}},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook\n%matplotlib inline \nimport cv2 as cv","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:55:57.191963Z","iopub.execute_input":"2021-08-13T21:55:57.192325Z","iopub.status.idle":"2021-08-13T21:56:00.254751Z","shell.execute_reply.started":"2021-08-13T21:55:57.192268Z","shell.execute_reply":"2021-08-13T21:56:00.253919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:00.257288Z","iopub.execute_input":"2021-08-13T21:56:00.257609Z","iopub.status.idle":"2021-08-13T21:56:00.623496Z","shell.execute_reply.started":"2021-08-13T21:56:00.257555Z","shell.execute_reply":"2021-08-13T21:56:00.622411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '../input/haar-cascades-for-face-detection'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:00.624775Z","iopub.execute_input":"2021-08-13T21:56:00.625063Z","iopub.status.idle":"2021-08-13T21:56:00.636857Z","shell.execute_reply.started":"2021-08-13T21:56:00.625006Z","shell.execute_reply":"2021-08-13T21:56:00.6361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")      ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:00.637911Z","iopub.execute_input":"2021-08-13T21:56:00.638274Z","iopub.status.idle":"2021-08-13T21:56:00.645902Z","shell.execute_reply.started":"2021-08-13T21:56:00.638219Z","shell.execute_reply":"2021-08-13T21:56:00.645116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file_ext in ext_dict:\n    print(f\"Files with extension `{file_ext}`: {len([file for file in train_list if  file.endswith(file_ext)])}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:02.203143Z","iopub.execute_input":"2021-08-13T21:56:02.203806Z","iopub.status.idle":"2021-08-13T21:56:02.210146Z","shell.execute_reply.started":"2021-08-13T21:56:02.203756Z","shell.execute_reply":"2021-08-13T21:56:02.209014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")\nfor file_ext in ext_dict:\n    print(f\"Files with extension `{file_ext}`: {len([file for file in train_list if  file.endswith(file_ext)])}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:04.162169Z","iopub.execute_input":"2021-08-13T21:56:04.162592Z","iopub.status.idle":"2021-08-13T21:56:04.172102Z","shell.execute_reply.started":"2021-08-13T21:56:04.162533Z","shell.execute_reply":"2021-08-13T21:56:04.170862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:09.417901Z","iopub.execute_input":"2021-08-13T21:56:09.418248Z","iopub.status.idle":"2021-08-13T21:56:09.424685Z","shell.execute_reply.started":"2021-08-13T21:56:09.418206Z","shell.execute_reply":"2021-08-13T21:56:09.423725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_meta_from_json(path):\n    df = pd.read_json(os.path.join(DATA_FOLDER, path, json_file))\n    df = df.T\n    return df\n\nmeta_train_df = get_meta_from_json(TRAIN_SAMPLE_FOLDER)\nmeta_train_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:11.040847Z","iopub.execute_input":"2021-08-13T21:56:11.041208Z","iopub.status.idle":"2021-08-13T21:56:11.855782Z","shell.execute_reply.started":"2021-08-13T21:56:11.041162Z","shell.execute_reply":"2021-08-13T21:56:11.854519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:18.208933Z","iopub.execute_input":"2021-08-13T21:56:18.209391Z","iopub.status.idle":"2021-08-13T21:56:18.221615Z","shell.execute_reply.started":"2021-08-13T21:56:18.209328Z","shell.execute_reply":"2021-08-13T21:56:18.220337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:22.023283Z","iopub.execute_input":"2021-08-13T21:56:22.023683Z","iopub.status.idle":"2021-08-13T21:56:22.156612Z","shell.execute_reply.started":"2021-08-13T21:56:22.023614Z","shell.execute_reply":"2021-08-13T21:56:22.155642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df.loc[meta_train_df.label=='REAL'])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:25.151102Z","iopub.execute_input":"2021-08-13T21:56:25.151474Z","iopub.status.idle":"2021-08-13T21:56:25.169907Z","shell.execute_reply.started":"2021-08-13T21:56:25.151406Z","shell.execute_reply":"2021-08-13T21:56:25.169248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:27.664932Z","iopub.execute_input":"2021-08-13T21:56:27.665474Z","iopub.status.idle":"2021-08-13T21:56:27.672309Z","shell.execute_reply.started":"2021-08-13T21:56:27.665398Z","shell.execute_reply":"2021-08-13T21:56:27.671098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(meta_train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:29.878806Z","iopub.execute_input":"2021-08-13T21:56:29.879336Z","iopub.status.idle":"2021-08-13T21:56:29.896062Z","shell.execute_reply.started":"2021-08-13T21:56:29.879272Z","shell.execute_reply":"2021-08-13T21:56:29.895444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_frequent_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    items = []\n    vals = []\n    for col in data.columns:\n        itm = data[col].value_counts().index[0]\n        val = data[col].value_counts().values[0]\n        items.append(itm)\n        vals.append(val)\n    tt['Most frequent item'] = items\n    tt['Frequence'] = vals\n    tt['Percent from total'] = np.round(vals / total * 100, 3)\n    return(np.transpose(tt))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:32.448772Z","iopub.execute_input":"2021-08-13T21:56:32.449201Z","iopub.status.idle":"2021-08-13T21:56:32.460248Z","shell.execute_reply.started":"2021-08-13T21:56:32.449126Z","shell.execute_reply":"2021-08-13T21:56:32.458734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:34.600186Z","iopub.execute_input":"2021-08-13T21:56:34.600654Z","iopub.status.idle":"2021-08-13T21:56:34.626357Z","shell.execute_reply.started":"2021-08-13T21:56:34.600601Z","shell.execute_reply":"2021-08-13T21:56:34.625707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n    '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n    '''\n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    g = sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n    g.set_title(\"Number and percentage of {}\".format(title))\n    if(size > 2):\n        plt.xticks(rotation=90, size=8)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:37.001262Z","iopub.execute_input":"2021-08-13T21:56:37.001652Z","iopub.status.idle":"2021-08-13T21:56:37.012565Z","shell.execute_reply.started":"2021-08-13T21:56:37.001589Z","shell.execute_reply":"2021-08-13T21:56:37.011319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('split', 'split (train)', meta_train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:39.288406Z","iopub.execute_input":"2021-08-13T21:56:39.288759Z","iopub.status.idle":"2021-08-13T21:56:39.511146Z","shell.execute_reply.started":"2021-08-13T21:56:39.288698Z","shell.execute_reply":"2021-08-13T21:56:39.509496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('label', 'label (train)', meta_train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:42.296045Z","iopub.execute_input":"2021-08-13T21:56:42.29658Z","iopub.status.idle":"2021-08-13T21:56:42.504327Z","shell.execute_reply.started":"2021-08-13T21:56:42.296527Z","shell.execute_reply":"2021-08-13T21:56:42.503168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = np.array(list(meta_train_df.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-13T21:56:45.87371Z","iopub.execute_input":"2021-08-13T21:56:45.874118Z","iopub.status.idle":"2021-08-13T21:56:45.886108Z","shell.execute_reply.started":"2021-08-13T21:56:45.874037Z","shell.execute_reply":"2021-08-13T21:56:45.884836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='FAKE'].sample(3).index)\nfake_train_sample_video","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:47.449681Z","iopub.execute_input":"2021-08-13T21:56:47.450265Z","iopub.status.idle":"2021-08-13T21:56:47.46074Z","shell.execute_reply.started":"2021-08-13T21:56:47.450214Z","shell.execute_reply":"2021-08-13T21:56:47.459629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_image = cv.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:49.328129Z","iopub.execute_input":"2021-08-13T21:56:49.328515Z","iopub.status.idle":"2021-08-13T21:56:49.337119Z","shell.execute_reply.started":"2021-08-13T21:56:49.328456Z","shell.execute_reply":"2021-08-13T21:56:49.335665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:56:51.40953Z","iopub.execute_input":"2021-08-13T21:56:51.409888Z","iopub.status.idle":"2021-08-13T21:56:53.54202Z","shell.execute_reply.started":"2021-08-13T21:56:51.409832Z","shell.execute_reply":"2021-08-13T21:56:53.540916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='REAL'].sample(3).index)\nreal_train_sample_video","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:57:03.559742Z","iopub.execute_input":"2021-08-13T21:57:03.560144Z","iopub.status.idle":"2021-08-13T21:57:03.569507Z","shell.execute_reply.started":"2021-08-13T21:57:03.560068Z","shell.execute_reply":"2021-08-13T21:57:03.568462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:57:05.436805Z","iopub.execute_input":"2021-08-13T21:57:05.437504Z","iopub.status.idle":"2021-08-13T21:57:07.417209Z","shell.execute_reply.started":"2021-08-13T21:57:05.437425Z","shell.execute_reply":"2021-08-13T21:57:07.416168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_train_df['original'].value_counts()[0:5]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:57:12.946899Z","iopub.execute_input":"2021-08-13T21:57:12.947245Z","iopub.status.idle":"2021-08-13T21:57:12.9573Z","shell.execute_reply.started":"2021-08-13T21:57:12.947193Z","shell.execute_reply":"2021-08-13T21:57:12.956141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    # we only show images extracted from the first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n        video_path = os.path.join(DATA_FOLDER, video_folder,video_file)\n        capture_image = cv.VideoCapture(video_path) \n        ret, frame = capture_image.read()\n        frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n        ax[i//3, i%3].imshow(frame)\n        ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n        ax[i//3, i%3].axis('on')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:57:15.313842Z","iopub.execute_input":"2021-08-13T21:57:15.314176Z","iopub.status.idle":"2021-08-13T21:57:15.32432Z","shell.execute_reply.started":"2021-08-13T21:57:15.314123Z","shell.execute_reply":"2021-08-13T21:57:15.323138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='meawmsgiti.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:57:17.966206Z","iopub.execute_input":"2021-08-13T21:57:17.966545Z","iopub.status.idle":"2021-08-13T21:57:20.772229Z","shell.execute_reply.started":"2021-08-13T21:57:17.966493Z","shell.execute_reply":"2021-08-13T21:57:20.771216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='atvmxvwyns.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:03.109757Z","iopub.execute_input":"2021-08-13T21:58:03.110362Z","iopub.status.idle":"2021-08-13T21:58:05.84304Z","shell.execute_reply.started":"2021-08-13T21:58:03.110123Z","shell.execute_reply":"2021-08-13T21:58:05.842312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='qeumxirsme.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:11.46888Z","iopub.execute_input":"2021-08-13T21:58:11.469434Z","iopub.status.idle":"2021-08-13T21:58:13.879145Z","shell.execute_reply.started":"2021-08-13T21:58:11.469385Z","shell.execute_reply":"2021-08-13T21:58:13.878057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='kgbkktcjxf.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:14.090152Z","iopub.execute_input":"2021-08-13T21:58:14.090555Z","iopub.status.idle":"2021-08-13T21:58:16.619516Z","shell.execute_reply.started":"2021-08-13T21:58:14.090492Z","shell.execute_reply":"2021-08-13T21:58:16.618716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:21.032723Z","iopub.execute_input":"2021-08-13T21:58:21.03342Z","iopub.status.idle":"2021-08-13T21:58:21.04213Z","shell.execute_reply.started":"2021-08-13T21:58:21.033364Z","shell.execute_reply":"2021-08-13T21:58:21.041346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:24.95732Z","iopub.execute_input":"2021-08-13T21:58:24.957783Z","iopub.status.idle":"2021-08-13T21:58:24.968507Z","shell.execute_reply.started":"2021-08-13T21:58:24.95774Z","shell.execute_reply":"2021-08-13T21:58:24.967581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_from_video(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[0].video))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:27.523411Z","iopub.execute_input":"2021-08-13T21:58:27.523763Z","iopub.status.idle":"2021-08-13T21:58:28.123663Z","shell.execute_reply.started":"2021-08-13T21:58:27.523707Z","shell.execute_reply":"2021-08-13T21:58:28.122621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_from_video_list(test_videos.sample(6).video, TEST_FOLDER)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:30.956269Z","iopub.execute_input":"2021-08-13T21:58:30.956897Z","iopub.status.idle":"2021-08-13T21:58:33.937944Z","shell.execute_reply.started":"2021-08-13T21:58:30.956846Z","shell.execute_reply":"2021-08-13T21:58:33.936764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ObjectDetector():\n    '''\n    Class for Object Detection\n    '''\n    def __init__(self,object_cascade_path):\n        '''\n        param: object_cascade_path - path for the *.xml defining the parameters for {face, eye, smile, profile}\n        detection algorithm\n        source of the haarcascade resource is: https://github.com/opencv/opencv/tree/master/data/haarcascades\n        '''\n\n        self.objectCascade=cv.CascadeClassifier(object_cascade_path)\n\n\n    def detect(self, image, scale_factor=1.3,\n               min_neighbors=5,\n               min_size=(20,20)):\n        '''\n        Function return rectangle coordinates of object for given image\n        param: image - image to process\n        param: scale_factor - scale factor used for object detection\n        param: min_neighbors - minimum number of parameters considered during object detection\n        param: min_size - minimum size of bounding box for object detected\n        '''\n        rects=self.objectCascade.detectMultiScale(image,\n                                                scaleFactor=scale_factor,\n                                                minNeighbors=min_neighbors,\n                                                minSize=min_size)\n        return rects","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:38.000985Z","iopub.execute_input":"2021-08-13T21:58:38.001383Z","iopub.status.idle":"2021-08-13T21:58:38.010659Z","shell.execute_reply.started":"2021-08-13T21:58:38.001324Z","shell.execute_reply":"2021-08-13T21:58:38.009215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Frontal face, profile, eye and smile  haar cascade loaded\nfrontal_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_frontalface_default.xml')\neye_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_eye.xml')\nprofile_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_profileface.xml')\nsmile_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_smile.xml')\n\n#Detector object created\n# frontal face\nfd=ObjectDetector(frontal_cascade_path)\n# eye\ned=ObjectDetector(eye_cascade_path)\n# profile face\npd=ObjectDetector(profile_cascade_path)\n# smile\nsd=ObjectDetector(smile_cascade_path)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:41.558125Z","iopub.execute_input":"2021-08-13T21:58:41.558486Z","iopub.status.idle":"2021-08-13T21:58:41.712234Z","shell.execute_reply.started":"2021-08-13T21:58:41.558428Z","shell.execute_reply":"2021-08-13T21:58:41.711411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def detect_objects(image, scale_factor, min_neighbors, min_size):\n    '''\n    Objects detection function\n    Identify frontal face, eyes, smile and profile face and display the detected objects over the image\n    param: image - the image extracted from the video\n    param: scale_factor - scale factor parameter for `detect` function of ObjectDetector object\n    param: min_neighbors - min neighbors parameter for `detect` function of ObjectDetector object\n    param: min_size - minimum size parameter for f`detect` function of ObjectDetector object\n    '''\n    \n    image_gray=cv.cvtColor(image, cv.COLOR_BGR2GRAY)\n\n\n    eyes=ed.detect(image_gray,\n                   scale_factor=scale_factor,\n                   min_neighbors=min_neighbors,\n                   min_size=(int(min_size[0]/2), int(min_size[1]/2)))\n\n    for x, y, w, h in eyes:\n        #detected eyes shown in color image\n        cv.circle(image,(int(x+w/2),int(y+h/2)),(int((w + h)/4)),(0, 0,255),3)\n \n    # deactivated due to many false positive\n    #smiles=sd.detect(image_gray,\n    #               scale_factor=scale_factor,\n    #               min_neighbors=min_neighbors,\n    #               min_size=(int(min_size[0]/2), int(min_size[1]/2)))\n\n    #for x, y, w, h in smiles:\n    #    #detected smiles shown in color image\n    #    cv.rectangle(image,(x,y),(x+w, y+h),(0, 0,255),3)\n\n\n    profiles=pd.detect(image_gray,\n                   scale_factor=scale_factor,\n                   min_neighbors=min_neighbors,\n                   min_size=min_size)\n\n    for x, y, w, h in profiles:\n        #detected profiles shown in color image\n        cv.rectangle(image,(x,y),(x+w, y+h),(255, 0,0),3)\n\n    faces=fd.detect(image_gray,\n                   scale_factor=scale_factor,\n                   min_neighbors=min_neighbors,\n                   min_size=min_size)\n\n    for x, y, w, h in faces:\n        #detected faces shown in color image\n        cv.rectangle(image,(x,y),(x+w, y+h),(0, 255,0),3)\n\n    # image\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    image = cv.cvtColor(image, cv.COLOR_BGR2RGB)\n    ax.imshow(image)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:44.871699Z","iopub.execute_input":"2021-08-13T21:58:44.872058Z","iopub.status.idle":"2021-08-13T21:58:44.887941Z","shell.execute_reply.started":"2021-08-13T21:58:44.872001Z","shell.execute_reply":"2021-08-13T21:58:44.886569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_image_objects(video_file, video_set_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    Extract one image from the video and then perform face/eyes/smile/profile detection on the image\n    param: video_file - the video from which to extract the image from which we extract the face\n    '''\n    video_path = os.path.join(DATA_FOLDER, video_set_folder,video_file)\n    capture_image = cv.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    #frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    detect_objects(image=frame, \n            scale_factor=1.3, \n            min_neighbors=5, \n            min_size=(50, 50))  \n  ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:53.917869Z","iopub.execute_input":"2021-08-13T21:58:53.918235Z","iopub.status.idle":"2021-08-13T21:58:53.926725Z","shell.execute_reply.started":"2021-08-13T21:58:53.918173Z","shell.execute_reply":"2021-08-13T21:58:53.925025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='kgbkktcjxf.mp4'].index)\nfor video_file in same_original_fake_train_sample_video[1:4]:\n    print(video_file)\n    extract_image_objects(video_file)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:58:58.969296Z","iopub.execute_input":"2021-08-13T21:58:58.969657Z","iopub.status.idle":"2021-08-13T21:59:01.273032Z","shell.execute_reply.started":"2021-08-13T21:58:58.969603Z","shell.execute_reply":"2021-08-13T21:59:01.272198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_subsample_video = list(meta_train_df.sample(3).index)\nfor video_file in train_subsample_video:\n    print(video_file)\n    extract_image_objects(video_file)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T21:59:05.053886Z","iopub.execute_input":"2021-08-13T21:59:05.054268Z","iopub.status.idle":"2021-08-13T21:59:08.222725Z","shell.execute_reply.started":"2021-08-13T21:59:05.054209Z","shell.execute_reply":"2021-08-13T21:59:08.221804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subsample_test_videos = list(test_videos.sample(3).video)\nfor video_file in subsample_test_videos:\n    print(video_file)\n    extract_image_objects(video_file, TEST_FOLDER)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:59:13.138215Z","iopub.execute_input":"2021-08-13T21:59:13.138585Z","iopub.status.idle":"2021-08-13T21:59:16.143865Z","shell.execute_reply.started":"2021-08-13T21:59:13.138518Z","shell.execute_reply":"2021-08-13T21:59:16.14248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_videos = list(meta_train_df.loc[meta_train_df.label=='FAKE'].index)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:59:20.279398Z","iopub.execute_input":"2021-08-13T21:59:20.279877Z","iopub.status.idle":"2021-08-13T21:59:20.286125Z","shell.execute_reply.started":"2021-08-13T21:59:20.279833Z","shell.execute_reply":"2021-08-13T21:59:20.284636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\n\ndef play_video(video_file, subset=TRAIN_SAMPLE_FOLDER):\n    '''\n    Display video\n    param: video_file - the name of the video file to display\n    param: subset - the folder where the video file is located (can be TRAIN_SAMPLE_FOLDER or TEST_Folder)\n    '''\n    video_url = open(os.path.join(DATA_FOLDER, subset,video_file),'rb').read()\n    data_url = \"data:video/mp4;base64,\" + b64encode(video_url).decode()\n    return HTML(\"\"\"<video width=500 controls><source src=\"%s\" type=\"video/mp4\"></video>\"\"\" % data_url)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:59:22.503282Z","iopub.execute_input":"2021-08-13T21:59:22.503631Z","iopub.status.idle":"2021-08-13T21:59:22.511399Z","shell.execute_reply.started":"2021-08-13T21:59:22.503574Z","shell.execute_reply":"2021-08-13T21:59:22.510401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[0])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T21:59:25.561583Z","iopub.execute_input":"2021-08-13T21:59:25.562022Z","iopub.status.idle":"2021-08-13T21:59:26.352553Z","shell.execute_reply.started":"2021-08-13T21:59:25.561955Z","shell.execute_reply":"2021-08-13T21:59:26.351472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[1])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-13T23:51:09.204152Z","iopub.execute_input":"2021-08-13T23:51:09.204479Z","iopub.status.idle":"2021-08-13T23:51:09.295189Z","shell.execute_reply.started":"2021-08-13T23:51:09.204421Z","shell.execute_reply":"2021-08-13T23:51:09.294101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[2])","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[3])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[4])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[5])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[10])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[12])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[15])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_video(fake_videos[18])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <a id=\"7\">Referências</a>\n\n[1] Deepfake, Wikipedia, https://en.wikipedia.org/wiki/Deepfake  \n[2] Google DeepFake Database, Endgadget, https://www.engadget.com/2019/09/25/google-deepfake-database/  \n[3] A quick look at the first frame of each video,  https://www.kaggle.com/brassmonkey381/a-quick-look-at-the-first-frame-of-each-video  \n[4] Basic EDA Face Detection, split video, ROI, https://www.kaggle.com/marcovasquez/basic-eda-face-detection-split-video-roi  \n[5] Face Detection with OpenCV, https://www.kaggle.com/serkanpeldek/face-detection-with-opencv   \n[6] Play video and processing, https://www.kaggle.com/hamditarek/play-video-and-processing/\n","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0"}}]}