{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\njson_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input/samplemegadata/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        json_files.append(os.path.join(dirname, filename))\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### This notebook simply analysis the summary of the whole dataset from the megadata json files"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import json as js\nimport pathlib\nfrom tqdm import tqdm\njson_files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fake_num = 0\nreal_num = 0\nfile_num = 0\nvideo_num = 0\nfake_real = 0\ntrain_num = 0\ntest_num = 0\nfake_files = []\nreal_files = []\nno_origin = []\nfor path in tqdm(json_files):    \n    with open(path, 'r') as f:\n        file_num += 1\n        j = js.loads(f.read())\n        video_num += len(j.keys())\n        for key in j.keys():\n            video = j[key]\n            if video['split'] == 'train':\n                train_num += 1\n            else:\n                test_num += 1\n\n            if video['label'] == 'FAKE':\n                fake_num += 1\n                fake_files.append(key)\n                if 'original' in video.keys():\n                    fake_real += 1\n                else:\n                    no_origin.append(video)\n            else:\n                real_num += 1\n                real_files.append(key)\n        \nprint('fake_num is: ', fake_num)\nprint('real_num is: ', real_num)\nprint('file_num is: ', file_num)\nprint('video_num is: ', video_num)\nprint('fake_real is: ', fake_real)\nprint('train_num is: ', train_num)\nprint('test_num is: ', test_num)\n# print('fake_files is: ', fake_files)\n# print('real_files is: ', real_files)\nprint('no_origin is: ', no_origin)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}