{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8041358,"sourceType":"datasetVersion","datasetId":4740976}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom glob import glob \npd.set_option('display.max_columns', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-06T03:30:32.020094Z","iopub.execute_input":"2024-04-06T03:30:32.020596Z","iopub.status.idle":"2024-04-06T03:30:32.027136Z","shell.execute_reply.started":"2024-04-06T03:30:32.020561Z","shell.execute_reply":"2024-04-06T03:30:32.025893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = glob(\"/kaggle/input/birdclef2024-metadataset/metadata_json/**/*.json\")\n\nuse_col = [\n    'id', \n    'gen', \n    'sp', \n    'ssp', \n    #'group', \n    #'en', \n    'rec', \n    #'cnt', \n    #'loc', \n    'lat',\n    'lng', \n    #'alt', \n    'type', \n    'sex', \n    'stage', \n    'method', \n    'url', \n    'file',\n    'file-name',\n    'lic', \n    'q', \n    'length', \n    'time', \n    'date',\n    'uploaded', \n    'also',\n    #'rmk', \n    'bird-seen', \n    'animal-seen', \n    'playback-used', \n    #'temp', \n    #'regnr',\n    #'auto', \n    #'dvc', \n    #'mic', \n    'smp', \n    #'sono.small', 'sono.med', 'sono.large',\n    #'sono.full', 'osci.small', 'osci.med', 'osci.large'\n]\n\ndf = []\nfor path in paths:\n    tmp = pd.read_json(path)\n    if len(tmp) > 0:\n        df.append(pd.json_normalize(tmp[\"recordings\"])[use_col])\ndf = pd.concat(df)\ndf.to_csv(\"metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-06T03:28:03.871044Z","iopub.execute_input":"2024-04-06T03:28:03.871557Z","iopub.status.idle":"2024-04-06T03:28:04.733145Z","shell.execute_reply.started":"2024-04-06T03:28:03.871508Z","shell.execute_reply":"2024-04-06T03:28:04.731924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-04-06T03:32:57.484787Z","iopub.execute_input":"2024-04-06T03:32:57.486015Z","iopub.status.idle":"2024-04-06T03:34:01.265759Z","shell.execute_reply.started":"2024-04-06T03:32:57.485963Z","shell.execute_reply":"2024-04-06T03:34:01.264291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-04-06T03:37:29.840223Z","iopub.execute_input":"2024-04-06T03:37:29.840741Z","iopub.status.idle":"2024-04-06T03:37:33.136297Z","shell.execute_reply.started":"2024-04-06T03:37:29.840705Z","shell.execute_reply":"2024-04-06T03:37:33.134969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}