{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Introduction\n\nThe various metainformation for this competition is stored in json format.\n\nWe would like to process these jsons so that we can easily build our training matrices.\n\nFor this, we will process all jsons and extract dataframes, by normalizing the json data."},{"metadata":{},"cell_type":"markdown","source":"# Data ingestion and processing\n\n\nWe will do all data ingestion and processing into a single loop."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport json","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"json_folder_path = \"/kaggle/input/iwildcam2021-fgvc8/metadata\"\nlist_of_files = list(os.listdir(json_folder_path))\n\nfor file_name in list_of_files:\n    json_path = os.path.join(json_folder_path, file_name)\n    print(f\"Current json processed: {file_name}\")\n    with open(json_path) as json_file:\n        # read each json\n        json_data = json.load(json_file)\n        # for each item in the json\n        for item in json_data.items():\n            # prepare the dataframe name\n            file_name_split = file_name.split(\".\")[0]\n            file_name_split = file_name_split.split(\"_\")\n            file_name_str = file_name_split[1] + \"_\" + file_name_split[2]\n            print(f\"\\tCurrent json item processed: {item[0]} length: {len(item[1])}\")\n            data_frame_name = f\"{file_name_str}_{item[0]}_df\"\n            print(f\"\\tDynamic dataframe created: {data_frame_name}\")\n            # dynamic creation of a dataframe, using vars()[data_frame_name]\n            vars()[data_frame_name] = pd.json_normalize(json_data.get(item[0]))\n            # output the dataframe\n            vars()[data_frame_name].to_csv(f\"{data_frame_name}\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(megadetector_results_images_df.shape)\nmegadetector_results_images_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's further process `megadetector_results_images_df.detections`\n\nLet's find what is the maximum number of  detections from all data."},{"metadata":{"trusted":true},"cell_type":"code","source":"megadetector_results_images_df['detections_count'] = megadetector_results_images_df[\"detections\"].apply(lambda x: len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"Max detections: {max(megadetector_results_images_df['detections_count'] )}\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We will keep this data in this format for now."},{"metadata":{"trusted":true},"cell_type":"code","source":"print(megadetector_results_info_df.shape)\nmegadetector_results_info_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(megadetector_results_detection_categories_df.shape)\nmegadetector_results_detection_categories_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(test_information_images_df.shape)\ntest_information_images_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_annotations_images_df.shape)\ntrain_annotations_images_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_annotations_annotations_df.shape)\ntrain_annotations_annotations_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_annotations_categories_df.shape)\ntrain_annotations_categories_df.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}