{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":10368024,"sourceType":"datasetVersion","datasetId":6421720}],"dockerImageVersionId":29844,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:46.740793Z","iopub.execute_input":"2025-01-04T07:07:46.741140Z","iopub.status.idle":"2025-01-04T07:07:47.289287Z","shell.execute_reply.started":"2025-01-04T07:07:46.741077Z","shell.execute_reply":"2025-01-04T07:07:47.288226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import requests\nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook\n%matplotlib inline \nimport cv2 as cv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.291685Z","iopub.execute_input":"2025-01-04T07:07:47.291964Z","iopub.status.idle":"2025-01-04T07:07:47.299921Z","shell.execute_reply.started":"2025-01-04T07:07:47.291917Z","shell.execute_reply":"2025-01-04T07:07:47.298475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FACE_DETECTION_FOLDER = '/kaggle/input/haarcascade'\nprint(f\"Face detection resources: {os.listdir(FACE_DETECTION_FOLDER)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.302443Z","iopub.execute_input":"2025-01-04T07:07:47.302859Z","iopub.status.idle":"2025-01-04T07:07:47.319909Z","shell.execute_reply.started":"2025-01-04T07:07:47.302784Z","shell.execute_reply":"2025-01-04T07:07:47.318279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.321574Z","iopub.execute_input":"2025-01-04T07:07:47.322052Z","iopub.status.idle":"2025-01-04T07:07:47.353224Z","shell.execute_reply.started":"2025-01-04T07:07:47.321907Z","shell.execute_reply":"2025-01-04T07:07:47.351899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_list = list(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))\next_dict = []\nfor file in train_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")      ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.361542Z","iopub.execute_input":"2025-01-04T07:07:47.361850Z","iopub.status.idle":"2025-01-04T07:07:47.373268Z","shell.execute_reply.started":"2025-01-04T07:07:47.361803Z","shell.execute_reply":"2025-01-04T07:07:47.371647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for file_ext in ext_dict:\n    print(f\"Files with extension `{file_ext}`: {len([file for file in train_list if  file.endswith(file_ext)])}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.382032Z","iopub.execute_input":"2025-01-04T07:07:47.382361Z","iopub.status.idle":"2025-01-04T07:07:47.388181Z","shell.execute_reply.started":"2025-01-04T07:07:47.382300Z","shell.execute_reply":"2025-01-04T07:07:47.387054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_list = list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))\next_dict = []\nfor file in test_list:\n    file_ext = file.split('.')[1]\n    if (file_ext not in ext_dict):\n        ext_dict.append(file_ext)\nprint(f\"Extensions: {ext_dict}\")\nfor file_ext in ext_dict:\n    print(f\"Files with extension `{file_ext}`: {len([file for file in train_list if  file.endswith(file_ext)])}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.390123Z","iopub.execute_input":"2025-01-04T07:07:47.390530Z","iopub.status.idle":"2025-01-04T07:07:47.410263Z","shell.execute_reply.started":"2025-01-04T07:07:47.390461Z","shell.execute_reply":"2025-01-04T07:07:47.409141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"json_file = [file for file in train_list if  file.endswith('json')][0]\nprint(f\"JSON file: {json_file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.411602Z","iopub.execute_input":"2025-01-04T07:07:47.411964Z","iopub.status.idle":"2025-01-04T07:07:47.419237Z","shell.execute_reply.started":"2025-01-04T07:07:47.411896Z","shell.execute_reply":"2025-01-04T07:07:47.418422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.421236Z","iopub.execute_input":"2025-01-04T07:07:47.421764Z","iopub.status.idle":"2025-01-04T07:07:47.626875Z","shell.execute_reply.started":"2025-01-04T07:07:47.421647Z","shell.execute_reply":"2025-01-04T07:07:47.625605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pylab as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.629195Z","iopub.execute_input":"2025-01-04T07:07:47.629600Z","iopub.status.idle":"2025-01-04T07:07:47.633977Z","shell.execute_reply.started":"2025-01-04T07:07:47.629526Z","shell.execute_reply":"2025-01-04T07:07:47.633099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.635256Z","iopub.execute_input":"2025-01-04T07:07:47.635544Z","iopub.status.idle":"2025-01-04T07:07:47.847483Z","shell.execute_reply.started":"2025-01-04T07:07:47.635492Z","shell.execute_reply":"2025-01-04T07:07:47.846534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.848867Z","iopub.execute_input":"2025-01-04T07:07:47.849201Z","iopub.status.idle":"2025-01-04T07:07:47.858683Z","shell.execute_reply.started":"2025-01-04T07:07:47.849152Z","shell.execute_reply":"2025-01-04T07:07:47.857683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data(train_sample_metadata)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.860159Z","iopub.execute_input":"2025-01-04T07:07:47.860640Z","iopub.status.idle":"2025-01-04T07:07:47.896405Z","shell.execute_reply.started":"2025-01-04T07:07:47.860588Z","shell.execute_reply":"2025-01-04T07:07:47.895126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data(train_sample_metadata.loc[train_sample_metadata.label=='REAL'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.897691Z","iopub.execute_input":"2025-01-04T07:07:47.897984Z","iopub.status.idle":"2025-01-04T07:07:47.921691Z","shell.execute_reply.started":"2025-01-04T07:07:47.897937Z","shell.execute_reply":"2025-01-04T07:07:47.920049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.925488Z","iopub.execute_input":"2025-01-04T07:07:47.926380Z","iopub.status.idle":"2025-01-04T07:07:47.937692Z","shell.execute_reply.started":"2025-01-04T07:07:47.926258Z","shell.execute_reply":"2025-01-04T07:07:47.935947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_values(train_sample_metadata)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.939575Z","iopub.execute_input":"2025-01-04T07:07:47.939973Z","iopub.status.idle":"2025-01-04T07:07:47.970550Z","shell.execute_reply.started":"2025-01-04T07:07:47.939901Z","shell.execute_reply":"2025-01-04T07:07:47.968539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.972376Z","iopub.execute_input":"2025-01-04T07:07:47.972762Z","iopub.status.idle":"2025-01-04T07:07:47.978711Z","shell.execute_reply.started":"2025-01-04T07:07:47.972691Z","shell.execute_reply":"2025-01-04T07:07:47.977529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n    '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n    '''\n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    g = sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n    g.set_title(\"Number and percentage of {}\".format(title))\n    if(size > 2):\n        plt.xticks(rotation=90, size=8)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.980593Z","iopub.execute_input":"2025-01-04T07:07:47.981162Z","iopub.status.idle":"2025-01-04T07:07:47.995586Z","shell.execute_reply.started":"2025-01-04T07:07:47.980926Z","shell.execute_reply":"2025-01-04T07:07:47.994397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_count('split', 'split (train)', train_sample_metadata)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:47.997224Z","iopub.execute_input":"2025-01-04T07:07:47.997636Z","iopub.status.idle":"2025-01-04T07:07:48.143508Z","shell.execute_reply.started":"2025-01-04T07:07:47.997563Z","shell.execute_reply":"2025-01-04T07:07:48.142561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_count('label', 'label (train)', train_sample_metadata)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:48.144820Z","iopub.execute_input":"2025-01-04T07:07:48.145115Z","iopub.status.idle":"2025-01-04T07:07:48.327538Z","shell.execute_reply.started":"2025-01-04T07:07:48.145068Z","shell.execute_reply":"2025-01-04T07:07:48.326731Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##### EDA","metadata":{}},{"cell_type":"code","source":"meta = np.array(list(train_sample_metadata.index))\nstorage = np.array([file for file in train_list if  file.endswith('mp4')])\nprint(f\"Metadata: {meta.shape[0]}, Folder: {storage.shape[0]}\")\nprint(f\"Files in metadata and not in folder: {np.setdiff1d(meta,storage,assume_unique=False).shape[0]}\")\nprint(f\"Files in folder and not in metadata: {np.setdiff1d(storage,meta,assume_unique=False).shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:48.328829Z","iopub.execute_input":"2025-01-04T07:07:48.329144Z","iopub.status.idle":"2025-01-04T07:07:48.339446Z","shell.execute_reply.started":"2025-01-04T07:07:48.329095Z","shell.execute_reply":"2025-01-04T07:07:48.338510Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Fake videos","metadata":{}},{"cell_type":"code","source":"import cv2 as cv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:48.340784Z","iopub.execute_input":"2025-01-04T07:07:48.341089Z","iopub.status.idle":"2025-01-04T07:07:48.361024Z","shell.execute_reply.started":"2025-01-04T07:07:48.341040Z","shell.execute_reply":"2025-01-04T07:07:48.359858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='FAKE'].sample(3).index)\nfake_train_sample_video","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:48.362863Z","iopub.execute_input":"2025-01-04T07:07:48.363205Z","iopub.status.idle":"2025-01-04T07:07:48.380919Z","shell.execute_reply.started":"2025-01-04T07:07:48.363155Z","shell.execute_reply":"2025-01-04T07:07:48.379901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def display_image_from_video(video_path):\n    capture_image = cv.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    ax.imshow(frame)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:48.382280Z","iopub.execute_input":"2025-01-04T07:07:48.382973Z","iopub.status.idle":"2025-01-04T07:07:48.396390Z","shell.execute_reply.started":"2025-01-04T07:07:48.382919Z","shell.execute_reply":"2025-01-04T07:07:48.395409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in fake_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:48.398777Z","iopub.execute_input":"2025-01-04T07:07:48.399132Z","iopub.status.idle":"2025-01-04T07:07:50.042363Z","shell.execute_reply.started":"2025-01-04T07:07:48.399079Z","shell.execute_reply":"2025-01-04T07:07:50.041707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Real videos","metadata":{}},{"cell_type":"code","source":"real_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.label=='REAL'].sample(3).index)\nreal_train_sample_video","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:50.043404Z","iopub.execute_input":"2025-01-04T07:07:50.043905Z","iopub.status.idle":"2025-01-04T07:07:50.052100Z","shell.execute_reply.started":"2025-01-04T07:07:50.043632Z","shell.execute_reply":"2025-01-04T07:07:50.051213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for video_file in real_train_sample_video:\n    display_image_from_video(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER, video_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:50.053885Z","iopub.execute_input":"2025-01-04T07:07:50.054214Z","iopub.status.idle":"2025-01-04T07:07:51.859856Z","shell.execute_reply.started":"2025-01-04T07:07:50.054150Z","shell.execute_reply":"2025-01-04T07:07:51.858811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Same original","metadata":{}},{"cell_type":"code","source":"train_sample_metadata['original'].value_counts()[0:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:51.861505Z","iopub.execute_input":"2025-01-04T07:07:51.861871Z","iopub.status.idle":"2025-01-04T07:07:51.873269Z","shell.execute_reply.started":"2025-01-04T07:07:51.861809Z","shell.execute_reply":"2025-01-04T07:07:51.872404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def display_image_from_video_list(video_path_list, video_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    input: video_path_list - path for video\n    process:\n    0. for each video in the video path list\n        1. perform a video capture from the video\n        2. read the image\n        3. display the image\n    '''\n    plt.figure()\n    fig, ax = plt.subplots(2,3,figsize=(16,8))\n    # we only show images extracted from the first 6 videos\n    for i, video_file in enumerate(video_path_list[0:6]):\n        video_path = os.path.join(DATA_FOLDER, video_folder,video_file)\n        capture_image = cv.VideoCapture(video_path) \n        ret, frame = capture_image.read()\n        frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n        ax[i//3, i%3].imshow(frame)\n        ax[i//3, i%3].set_title(f\"Video: {video_file}\")\n        ax[i//3, i%3].axis('on')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:51.874967Z","iopub.execute_input":"2025-01-04T07:07:51.875501Z","iopub.status.idle":"2025-01-04T07:07:51.888464Z","shell.execute_reply.started":"2025-01-04T07:07:51.875222Z","shell.execute_reply":"2025-01-04T07:07:51.887284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='atvmxvwyns.mp4'].index)\ndisplay_image_from_video_list(same_original_fake_train_sample_video)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:51.889976Z","iopub.execute_input":"2025-01-04T07:07:51.890285Z","iopub.status.idle":"2025-01-04T07:07:54.146969Z","shell.execute_reply.started":"2025-01-04T07:07:51.890229Z","shell.execute_reply":"2025-01-04T07:07:54.145828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_videos = pd.DataFrame(list(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER))), columns=['video'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.148680Z","iopub.execute_input":"2025-01-04T07:07:54.149075Z","iopub.status.idle":"2025-01-04T07:07:54.158647Z","shell.execute_reply.started":"2025-01-04T07:07:54.148986Z","shell.execute_reply":"2025-01-04T07:07:54.157543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_videos.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.163770Z","iopub.execute_input":"2025-01-04T07:07:54.164093Z","iopub.status.idle":"2025-01-04T07:07:54.176732Z","shell.execute_reply.started":"2025-01-04T07:07:54.164016Z","shell.execute_reply":"2025-01-04T07:07:54.175831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_image_from_video(os.path.join(DATA_FOLDER, TEST_FOLDER, test_videos.iloc[0].video))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.178389Z","iopub.execute_input":"2025-01-04T07:07:54.178873Z","iopub.status.idle":"2025-01-04T07:07:54.754686Z","shell.execute_reply.started":"2025-01-04T07:07:54.178678Z","shell.execute_reply":"2025-01-04T07:07:54.753631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of test videos: {test_videos.shape[0]}\")\nprint(f\"Example test videos: {test_videos.head()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.756305Z","iopub.execute_input":"2025-01-04T07:07:54.756621Z","iopub.status.idle":"2025-01-04T07:07:54.763681Z","shell.execute_reply.started":"2025-01-04T07:07:54.756572Z","shell.execute_reply":"2025-01-04T07:07:54.762460Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ----------","metadata":{}},{"cell_type":"markdown","source":"# feature engineering","metadata":{}},{"cell_type":"code","source":"class ObjectDetector():\n    '''\n    Class for Object Detection\n    '''\n    def __init__(self,object_cascade_path):\n        '''\n        param: object_cascade_path - path for the *.xml defining the parameters for {face, eye, smile, profile}\n        detection algorithm\n        source of the haarcascade resource is: https://github.com/opencv/opencv/tree/master/data/haarcascades\n        '''\n\n        self.objectCascade=cv.CascadeClassifier(object_cascade_path)\n\n\n    def detect(self, image, scale_factor=1.3,\n               min_neighbors=5,\n               min_size=(20,20)):\n        '''\n        Function return rectangle coordinates of object for given image\n        param: image - image to process\n        param: scale_factor - scale factor used for object detection\n        param: min_neighbors - minimum number of parameters considered during object detection\n        param: min_size - minimum size of bounding box for object detected\n        '''\n        rects=self.objectCascade.detectMultiScale(image,\n                                                scaleFactor=scale_factor,\n                                                minNeighbors=min_neighbors,\n                                                minSize=min_size)\n        return rects","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.764929Z","iopub.execute_input":"2025-01-04T07:07:54.765221Z","iopub.status.idle":"2025-01-04T07:07:54.779644Z","shell.execute_reply.started":"2025-01-04T07:07:54.765175Z","shell.execute_reply":"2025-01-04T07:07:54.778759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Frontal face, profile, eye and smile  haar cascade loaded\nfrontal_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_frontalface_default.xml')\neye_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_eye.xml')\nprofile_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_profileface.xml')\nsmile_cascade_path= os.path.join(FACE_DETECTION_FOLDER,'haarcascade_smile.xml')\n\n#Detector object created\n# frontal face\nfd=ObjectDetector(frontal_cascade_path)\n# eye\ned=ObjectDetector(eye_cascade_path)\n# profile face\npd=ObjectDetector(profile_cascade_path)\n# smile\nsd=ObjectDetector(smile_cascade_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.781532Z","iopub.execute_input":"2025-01-04T07:07:54.781867Z","iopub.status.idle":"2025-01-04T07:07:54.876988Z","shell.execute_reply.started":"2025-01-04T07:07:54.781816Z","shell.execute_reply":"2025-01-04T07:07:54.876189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_objects(image, scale_factor, min_neighbors, min_size):\n    '''\n    Objects detection function\n    Identify frontal face, eyes, smile and profile face and display the detected objects over the image\n    param: image - the image extracted from the video\n    param: scale_factor - scale factor parameter for `detect` function of ObjectDetector object\n    param: min_neighbors - min neighbors parameter for `detect` function of ObjectDetector object\n    param: min_size - minimum size parameter for f`detect` function of ObjectDetector object\n    '''\n    \n    image_gray=cv.cvtColor(image, cv.COLOR_BGR2GRAY)\n\n\n    eyes=ed.detect(image_gray,\n                   scale_factor=scale_factor,\n                   min_neighbors=min_neighbors,\n                   min_size=(int(min_size[0]/2), int(min_size[1]/2)))\n\n    for x, y, w, h in eyes:\n        #detected eyes shown in color image\n        cv.circle(image,(int(x+w/2),int(y+h/2)),(int((w + h)/4)),(0, 0,255),3)\n \n    # deactivated due to many false positive\n    #smiles=sd.detect(image_gray,\n    #               scale_factor=scale_factor,\n    #               min_neighbors=min_neighbors,\n    #               min_size=(int(min_size[0]/2), int(min_size[1]/2)))\n\n    #for x, y, w, h in smiles:\n    #    #detected smiles shown in color image\n    #    cv.rectangle(image,(x,y),(x+w, y+h),(0, 0,255),3)\n\n\n    profiles=pd.detect(image_gray,\n                   scale_factor=scale_factor,\n                   min_neighbors=min_neighbors,\n                   min_size=min_size)\n\n    for x, y, w, h in profiles:\n        #detected profiles shown in color image\n        cv.rectangle(image,(x,y),(x+w, y+h),(255, 0,0),3)\n\n    faces=fd.detect(image_gray,\n                   scale_factor=scale_factor,\n                   min_neighbors=min_neighbors,\n                   min_size=min_size)\n\n    for x, y, w, h in faces:\n        #detected faces shown in color image\n        cv.rectangle(image,(x,y),(x+w, y+h),(0, 255,0),3)\n\n    # image\n    fig = plt.figure(figsize=(10,10))\n    ax = fig.add_subplot(111)\n    image = cv.cvtColor(image, cv.COLOR_BGR2RGB)\n    ax.imshow(image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.878267Z","iopub.execute_input":"2025-01-04T07:07:54.878564Z","iopub.status.idle":"2025-01-04T07:07:54.891203Z","shell.execute_reply.started":"2025-01-04T07:07:54.878511Z","shell.execute_reply":"2025-01-04T07:07:54.890340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_image_objects(video_file, video_set_folder=TRAIN_SAMPLE_FOLDER):\n    '''\n    Extract one image from the video and then perform face/eyes/smile/profile detection on the image\n    param: video_file - the video from which to extract the image from which we extract the face\n    '''\n    video_path = os.path.join(DATA_FOLDER, video_set_folder,video_file)\n    capture_image = cv.VideoCapture(video_path) \n    ret, frame = capture_image.read()\n    #frame = cv.cvtColor(frame, cv.COLOR_BGR2RGB)\n    detect_objects(image=frame, \n            scale_factor=1.3, \n            min_neighbors=5, \n            min_size=(50, 50))  \n  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.892766Z","iopub.execute_input":"2025-01-04T07:07:54.893098Z","iopub.status.idle":"2025-01-04T07:07:54.913890Z","shell.execute_reply.started":"2025-01-04T07:07:54.893042Z","shell.execute_reply":"2025-01-04T07:07:54.912623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## --------------","metadata":{}},{"cell_type":"markdown","source":"### ---------------","metadata":{}},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='kgbkktcjxf.mp4'].index)\nfor video_file in same_original_fake_train_sample_video[1:4]:\n    print(video_file)\n    extract_image_objects(video_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:54.915268Z","iopub.execute_input":"2025-01-04T07:07:54.915610Z","iopub.status.idle":"2025-01-04T07:07:57.068063Z","shell.execute_reply.started":"2025-01-04T07:07:54.915556Z","shell.execute_reply":"2025-01-04T07:07:57.066981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"same_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='qtnjyomzwo.mp4'].index)\nfor video_file in same_original_fake_train_sample_video[1:4]:\n    print(video_file)\n    extract_image_objects(video_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T07:07:57.069528Z","iopub.execute_input":"2025-01-04T07:07:57.069903Z","iopub.status.idle":"2025-01-04T07:07:59.494248Z","shell.execute_reply.started":"2025-01-04T07:07:57.069844Z","shell.execute_reply":"2025-01-04T07:07:59.493055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport pandas as pd\nimport os\n\nclass ImageObjectExtractor:\n    def __init__(self, video_file):\n        # Initialize the video file and feature storage\n        self.video_file = video_file\n        self.features = {\n            \"eyes\": 0,     # Will store whether eyes were detected\n            \"profile\": 0,  # Will store whether profile was detected\n            \"smile\": 0,    # Will store whether smile was detected\n            \"head\": 0,     # Will store whether head (face) was detected\n        }\n\n        # Load pre-trained Haar cascade classifiers\n        self.face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_frontalface_default.xml')\n        self.eye_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_eye.xml')\n        self.profile_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_profileface.xml')\n        self.smile_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_smile.xml')\n\n    def extract_objects(self):\n        # Load the video using OpenCV\n        video_capture = cv2.VideoCapture(self.video_file)\n        \n        # Process the video frames (for simplicity, just use the first frame here)\n        ret, frame = video_capture.read()\n        if not ret:\n            print(\"Error reading video file.\")\n            return\n\n        # Convert the frame to grayscale for detection\n        gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n\n        # Detect faces (head) in the image\n        faces = self.face_cascade.detectMultiScale(gray, 1.1, 4)\n        print(faces)\n        if len(faces) > 0:\n            self.features[\"head\"] = 1  # Head detected\n        else:\n            self.features[\"head\"] = 0  # No head detected\n\n        # Detect eyes in the image\n        eyes = self.eye_cascade.detectMultiScale(gray)\n        if len(eyes) > 0:\n            self.features[\"eyes\"] = 1  # Eyes detected\n        else:\n            self.features[\"eyes\"] = 0  # No eyes detected\n\n        # Detect profile faces in the image\n        profiles = self.profile_cascade.detectMultiScale(gray)\n        if len(profiles) > 0:\n            self.features[\"profile\"] = 1  # Profile face detected\n        else:\n            self.features[\"profile\"] = 0  # No profile detected\n\n        # Detect smiles in the image\n        smiles = self.smile_cascade.detectMultiScale(gray)\n        if len(smiles) > 0:\n            self.features[\"smile\"] = 1  # Smile detected\n        else:\n            self.features[\"smile\"] = 0  # No smile detected\n\n        # Release the video capture object\n        video_capture.release()\n\n    def get_feature_data(self):\n        # Return the extracted feature data, including the video file name\n        return {\"video_file\": self.video_file, **self.features}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T08:10:43.201201Z","iopub.execute_input":"2025-01-04T08:10:43.201541Z","iopub.status.idle":"2025-01-04T08:10:43.215736Z","shell.execute_reply.started":"2025-01-04T08:10:43.201491Z","shell.execute_reply":"2025-01-04T08:10:43.214848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_data_list = []\n\nsame_original_fake_train_sample_video = list(train_sample_metadata.loc[train_sample_metadata.original=='qtnjyomzwo.mp4'].index)\nfor video_file in same_original_fake_train_sample_video[1:4]:\n    print(video_file)\n    extractor = ImageObjectExtractor(video_file)\n    \n    # Get the feature data (can be stored in a table or dataframe for further analysis)\n    feature_data = extractor.get_feature_data()\n    feature_data_list.append(feature_data)\n\ndf = pd.DataFrame(feature_data_list)\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T08:10:45.880115Z","iopub.execute_input":"2025-01-04T08:10:45.880465Z","iopub.status.idle":"2025-01-04T08:10:46.117918Z","shell.execute_reply.started":"2025-01-04T08:10:45.880404Z","shell.execute_reply":"2025-01-04T08:10:46.116850Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Machine learning","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\n# Example: Assuming you want to predict 'profile_detected' (binary classification)\n# Prepare the data\nX = df.drop(columns=[\"video_file\", \"profile\"])  # Features (eyes, nose, mouth, head, etc.)\ny = df[\"profile\"]  # Target variable (whether profile is detected)\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Train a Random Forest classifier\nmodel = RandomForestClassifier()\nmodel.fit(X_train, y_train)\n\n# Predict and evaluate the model\ny_pred = model.predict(X_test)\nprint(f\"Accuracy: {accuracy_score(y_test, y_pred)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T08:02:19.342003Z","iopub.execute_input":"2025-01-04T08:02:19.342360Z","iopub.status.idle":"2025-01-04T08:02:19.380962Z","shell.execute_reply.started":"2025-01-04T08:02:19.342307Z","shell.execute_reply":"2025-01-04T08:02:19.380029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}