{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-08T08:40:06.465355Z","iopub.execute_input":"2022-04-08T08:40:06.465972Z","iopub.status.idle":"2022-04-08T08:40:09.447261Z","shell.execute_reply.started":"2022-04-08T08:40:06.465881Z","shell.execute_reply":"2022-04-08T08:40:09.445866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/birdclef-2022/train_metadata.csv\")\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:09.449406Z","iopub.execute_input":"2022-04-08T08:40:09.449781Z","iopub.status.idle":"2022-04-08T08:40:09.596464Z","shell.execute_reply.started":"2022-04-08T08:40:09.449739Z","shell.execute_reply":"2022-04-08T08:40:09.595337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:09.59754Z","iopub.execute_input":"2022-04-08T08:40:09.597792Z","iopub.status.idle":"2022-04-08T08:40:11.069771Z","shell.execute_reply.started":"2022-04-08T08:40:09.597765Z","shell.execute_reply":"2022-04-08T08:40:11.068645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prm_lbl_cnt = train_data.groupby('primary_label')['primary_label'].count().reset_index(name = 'cnt')\nfig = px.bar(prm_lbl_cnt, x='cnt', y='primary_label', width=800, height=1400)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:11.07227Z","iopub.execute_input":"2022-04-08T08:40:11.072808Z","iopub.status.idle":"2022-04-08T08:40:12.188407Z","shell.execute_reply.started":"2022-04-08T08:40:11.072764Z","shell.execute_reply":"2022-04-08T08:40:12.187556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.type.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:12.189721Z","iopub.execute_input":"2022-04-08T08:40:12.189946Z","iopub.status.idle":"2022-04-08T08:40:12.203458Z","shell.execute_reply.started":"2022-04-08T08:40:12.189919Z","shell.execute_reply":"2022-04-08T08:40:12.202544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\ncount  = train_data['type'].explode().value_counts()#pd.Series(train_data['type'].str.replace('[\\[\\]\\']','').str.split(',').map(Counter).sum())\ncount","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:12.204481Z","iopub.execute_input":"2022-04-08T08:40:12.204744Z","iopub.status.idle":"2022-04-08T08:40:12.218368Z","shell.execute_reply.started":"2022-04-08T08:40:12.204713Z","shell.execute_reply":"2022-04-08T08:40:12.217412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#.apply(pd.Series).stack().value_counts()\n\nfrom itertools import chain\nfrom collections import Counter\n\npd.DataFrame.from_dict(Counter(chain(+train_data['type'])), orient='index').sort_values(0, ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:12.220527Z","iopub.execute_input":"2022-04-08T08:40:12.221114Z","iopub.status.idle":"2022-04-08T08:40:12.242129Z","shell.execute_reply.started":"2022-04-08T08:40:12.221067Z","shell.execute_reply":"2022-04-08T08:40:12.241492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.__version__","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:12.243049Z","iopub.execute_input":"2022-04-08T08:40:12.243758Z","iopub.status.idle":"2022-04-08T08:40:12.250165Z","shell.execute_reply.started":"2022-04-08T08:40:12.243723Z","shell.execute_reply":"2022-04-08T08:40:12.249145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_data.type[1])","metadata":{"execution":{"iopub.status.busy":"2022-04-08T08:40:12.251833Z","iopub.execute_input":"2022-04-08T08:40:12.252599Z","iopub.status.idle":"2022-04-08T08:40:12.26279Z","shell.execute_reply.started":"2022-04-08T08:40:12.252527Z","shell.execute_reply":"2022-04-08T08:40:12.261918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}