{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-09T07:34:24.890067Z","iopub.execute_input":"2021-10-09T07:34:24.890436Z","iopub.status.idle":"2021-10-09T07:34:25.139879Z","shell.execute_reply.started":"2021-10-09T07:34:24.890385Z","shell.execute_reply":"2021-10-09T07:34:25.139165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook\n%matplotlib inline \nimport cv2 as cv","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:25.145581Z","iopub.execute_input":"2021-10-09T07:34:25.145811Z","iopub.status.idle":"2021-10-09T07:34:27.344385Z","shell.execute_reply.started":"2021-10-09T07:34:25.14577Z","shell.execute_reply":"2021-10-09T07:34:27.343653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.347821Z","iopub.execute_input":"2021-10-09T07:34:27.348199Z","iopub.status.idle":"2021-10-09T07:34:27.357967Z","shell.execute_reply.started":"2021-10-09T07:34:27.348015Z","shell.execute_reply":"2021-10-09T07:34:27.357153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '../input/haarcascades'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:46:12.272674Z","iopub.status.idle":"2021-10-09T07:46:12.273441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install face-recognition","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:41:48.179922Z","iopub.execute_input":"2021-10-09T07:41:48.180245Z","iopub.status.idle":"2021-10-09T07:46:12.253844Z","shell.execute_reply.started":"2021-10-09T07:41:48.180194Z","shell.execute_reply":"2021-10-09T07:46:12.252936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")  ","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.487209Z","iopub.status.idle":"2021-10-09T07:34:27.487716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file_ext in ext_dict:\n    print(f\"Files with extension `{file_ext}`: {len([file for file in train_list if  file.endswith(file_ext)])}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.489527Z","iopub.status.idle":"2021-10-09T07:34:27.490209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")\nfor file_ext in ext_dict:\n    print(f\"Files with extension `{file_ext}`: {len([file for file in train_list if  file.endswith(file_ext)])}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.491483Z","iopub.status.idle":"2021-10-09T07:34:27.492152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.49344Z","iopub.status.idle":"2021-10-09T07:34:27.494131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_meta_from_json(path):\n    df = pd.read_json(os.path.join(DATA_FOLDER, path, json_file))\n    df = df.T\n    return df\n\nmeta_train_df = get_meta_from_json(TRAIN_SAMPLE_FOLDER)\nmeta_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.495416Z","iopub.status.idle":"2021-10-09T07:34:27.496168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.497385Z","iopub.status.idle":"2021-10-09T07:34:27.498024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.499233Z","iopub.status.idle":"2021-10-09T07:34:27.499901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data(meta_train_df.loc[meta_train_df.label=='REAL'])","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.501111Z","iopub.status.idle":"2021-10-09T07:34:27.501791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.503006Z","iopub.status.idle":"2021-10-09T07:34:27.503671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.504906Z","iopub.status.idle":"2021-10-09T07:34:27.505583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def most_frequent_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    items = []\n    vals = []\n    for col in data.columns:\n        itm = data[col].value_counts().index[0]\n        val = data[col].value_counts().values[0]\n        items.append(itm)\n        vals.append(val)\n    tt['Most frequent item'] = items\n    tt['Frequence'] = vals\n    tt['Percent from total'] = np.round(vals / total * 100, 3)\n    return(np.transpose(tt))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.506854Z","iopub.status.idle":"2021-10-09T07:34:27.507603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_frequent_values(meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.508843Z","iopub.status.idle":"2021-10-09T07:34:27.509522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n    '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n    '''\n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    g = sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n    g.set_title(\"Number and percentage of {}\".format(title))\n    if(size > 2):\n        plt.xticks(rotation=90, size=8)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()  ","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.51072Z","iopub.status.idle":"2021-10-09T07:34:27.511457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('split', 'split (train)', meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.512662Z","iopub.status.idle":"2021-10-09T07:34:27.513331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count('label', 'label (train)', meta_train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.514517Z","iopub.status.idle":"2021-10-09T07:34:27.515191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, the `REAL` are only 19.25% in train sample videos, with the `FAKE`s acounting for 80.75% of the samples. \n\n\n# <a id=\"4\">Video data exploration</a>\n\n\nIn the following we will explore some of the video data. \n\n\n## Missing video (or meta) data\n\nWe check first if the list of files in the meta info and the list from the folder are the same.\n\n","metadata":{}},{"cell_type":"code","source":"meta = np.array(list(meta_train_df.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.516554Z","iopub.status.idle":"2021-10-09T07:34:27.51722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='FAKE'].sample(3).index)\nfake_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.518391Z","iopub.status.idle":"2021-10-09T07:34:27.519059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    '''\n    input: video_path - path for video\n    process:\n    1. perform a video capture from the video\n    2. read the image\n    3. display the image\n    '''\n    capture_image = cv.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.520357Z","iopub.status.idle":"2021-10-09T07:34:27.521001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.522167Z","iopub.status.idle":"2021-10-09T07:34:27.522815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_train_sample_video = list(meta_train_df.loc[meta_train_df.label=='REAL'].sample(3).index)\nreal_train_sample_video","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.523987Z","iopub.status.idle":"2021-10-09T07:34:27.524734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.525933Z","iopub.status.idle":"2021-10-09T07:34:27.526582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_train_df['original'].value_counts()[0:5]","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.527825Z","iopub.status.idle":"2021-10-09T07:34:27.528574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    # we only show images extracted from the first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n        video_path = os.path.join(DATA_FOLDER, video_folder,video_file)\n        capture_image = cv.VideoCapture(video_path) \n        ret, frame = capture_image.read()\n        frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n        ax[i//3, i%3].imshow(frame)\n        ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n        ax[i//3, i%3].axis('on')","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.529722Z","iopub.status.idle":"2021-10-09T07:34:27.530376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='meawmsgiti.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.531657Z","iopub.status.idle":"2021-10-09T07:34:27.532331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='atvmxvwyns.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.533483Z","iopub.status.idle":"2021-10-09T07:34:27.534148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='qeumxirsme.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.535359Z","iopub.status.idle":"2021-10-09T07:34:27.536004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(meta_train_df.loc[meta_train_df.original=='kgbkktcjxf.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.537177Z","iopub.status.idle":"2021-10-09T07:34:27.537834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.539095Z","iopub.status.idle":"2021-10-09T07:34:27.539773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_videos.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.540938Z","iopub.status.idle":"2021-10-09T07:34:27.541647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_from_video(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[0].video))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.542824Z","iopub.status.idle":"2021-10-09T07:34:27.543469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_image_from_video_list(test_videos.sample(6).video, TEST_FOLDER)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.544609Z","iopub.status.idle":"2021-10-09T07:34:27.54531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <a id='5'>Face detection</a>  \n\nFrom [5] ([Face Detection using OpenCV](https://www.kaggle.com/serkanpeldek/face-detection-with-opencv)) by [@serkanpeldek](https://www.kaggle.com/serkanpeldek) we got and slightly modified the functions to extract face, profile face, eyes and smile.  \n\nThe class ObjectDetector initialize the cascade classifier (using the imported resource). The function **detect** uses a method of the CascadeClassifier to detect objects into images - in this case the face, eye, smile or profile face.","metadata":{}},{"cell_type":"code","source":"class ObjectDetector():\n    '''\n    Class for Object Detection\n    '''\n    def __init__(self,object_cascade_path):\n        '''\n        param: object_cascade_path - path for the *.xml defining the parameters for {face, eye, smile, profile}\n        detection algorithm\n        source of the haarcascade resource is: https://github.com/opencv/opencv/tree/master/data/haarcascades\n        '''\n\n        self.objectCascade=cv.CascadeClassifier(object_cascade_path)\n\n\n    def detect(self, image, scale_factor=1.3,\n               min_neighbors=5,\n               min_size=(20,20)):\n        '''\n        Function return rectangle coordinates of object for given image\n        param: image - image to process\n        param: scale_factor - scale factor used for object detection\n        param: min_neighbors - minimum number of parameters considered during object detection\n        param: min_size - minimum size of bounding box for object detected\n        '''\n        rects=self.objectCascade.detectMultiScale(image,\n                                                scaleFactor=scale_factor,\n                                                minNeighbors=min_neighbors,\n                                                minSize=min_size)\n        return rects","metadata":{"execution":{"iopub.status.busy":"2021-10-09T07:34:27.546513Z","iopub.status.idle":"2021-10-09T07:34:27.547176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}