{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"path = '/kaggle/input/deepfake-detection-challenge'\n# reading file names\ntrain_files = os.listdir(os.path.join(path, 'train_sample_videos'))\ntrain_files.remove('metadata.json')\ntest_files = os.listdir(os.path.join(path, 'test_videos'))\n\nprint(f'Number of Train files: {len(train_files)}\\nNumber of Test files: {len(test_files)}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Equal number of train and test samples","attachments":{}},{"metadata":{"trusted":true},"cell_type":"code","source":"# reading the labels json file\nlabels_df = pd.read_json(os.path.join(path, 'train_sample_videos/metadata.json'))\nlabels_df = labels_df.T\nprint(labels_df.shape)\nlabels_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# number of real and fake samples\nlabels_df['label'].value_counts(normalize=True)*100","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## The training data is skewed"},{"metadata":{"trusted":true},"cell_type":"code","source":"# gets the frame size for a video\ndef get_frame_size(file):\n    cap = cv2.VideoCapture(file)\n    ret, frame = cap.read()\n    #plt.imshow(frame)\n    shape = frame.shape\n    cap.release()\n    return shape\n\n# gets the fps and duration of video\ndef get_video_length(file):\n    cap = cv2.VideoCapture(file)\n    fps = cap.get(cv2.CAP_PROP_FPS)\n    frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    duration = frame_count/fps\n    cap.release()\n    return round(fps), round(duration)\n\n# extract metadata of all files and return in a dataframe\ndef extract_metadata(files, path):\n    frame_size_list = []\n    fps_list = []\n    duration_list = []\n    for i in tqdm(files):\n        shape = get_frame_size(os.path.join(path,f'{i}'))\n        fps, duration = get_video_length(os.path.join(path,f'{i}'))\n        frame_size_list.append(shape)\n        fps_list.append(fps)\n        duration_list.append(duration)\n\n    meta_df = pd.DataFrame(data={'frame_shape':frame_size_list, 'fps':fps_list, 'duration':duration_list}, index=files)\n    return meta_df\n\nprint(get_frame_size(os.path.join(path, 'train_sample_videos/aagfhgtpmv.mp4')))\nprint(get_video_length(os.path.join(path, 'train_sample_videos/aagfhgtpmv.mp4')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# getting metadata for train files\ntrain_meta = extract_metadata(train_files, os.path.join(path, 'train_sample_videos'))\ntrain_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_meta.frame_shape.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_meta.fps.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Duration in seconds')\nprint(train_meta.duration.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# getting metadata for test files\ntest_meta = extract_metadata(test_files, os.path.join(path, 'test_videos'))\ntest_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_meta.frame_shape.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_meta.fps.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_meta.duration.value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* All videos are of 10 seconds and in 30 fps\n* Videos are in 2 frame sizes:\n  * 1080 X 1920\n  * 1920 X 1080"},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv(f\"{path}/sample_submission.csv\")\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['label'] = 0.7\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}