{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T17:25:14.440491Z","iopub.execute_input":"2023-04-19T17:25:14.441647Z","iopub.status.idle":"2023-04-19T17:25:14.447155Z","shell.execute_reply.started":"2023-04-19T17:25:14.441602Z","shell.execute_reply":"2023-04-19T17:25:14.446032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## imports","metadata":{}},{"cell_type":"code","source":"from IPython.display import Audio\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport plotly.express as px \nfrom collections import Counter\nimport itertools\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:25:11.629959Z","iopub.execute_input":"2023-04-19T17:25:11.630973Z","iopub.status.idle":"2023-04-19T17:25:14.124046Z","shell.execute_reply.started":"2023-04-19T17:25:11.630911Z","shell.execute_reply":"2023-04-19T17:25:14.122733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## reading data","metadata":{}},{"cell_type":"code","source":"TEST_DATA_DIR = \"/kaggle/input/birdclef-2023/test_soundscapes\"\nMETA_DATA_PATH = \"/kaggle/input/birdclef-2023/train_metadata.csv\"\nTRAIN_DATA_DIR = \"/kaggle/input/birdclef-2023/train_audio\"","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.391411Z","iopub.execute_input":"2023-04-19T12:38:13.392555Z","iopub.status.idle":"2023-04-19T12:38:13.397725Z","shell.execute_reply.started":"2023-04-19T12:38:13.39249Z","shell.execute_reply":"2023-04-19T12:38:13.396103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta_data = pd.read_csv(META_DATA_PATH)\ndf_meta_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.400706Z","iopub.execute_input":"2023-04-19T12:38:13.401139Z","iopub.status.idle":"2023-04-19T12:38:13.566232Z","shell.execute_reply.started":"2023-04-19T12:38:13.401098Z","shell.execute_reply":"2023-04-19T12:38:13.564762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta_data['filename'] = df_meta_data['filename'].apply(lambda x: os.path.join(TRAIN_DATA_DIR,x) )\ndf_meta_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.567769Z","iopub.execute_input":"2023-04-19T12:38:13.56817Z","iopub.status.idle":"2023-04-19T12:38:13.626433Z","shell.execute_reply.started":"2023-04-19T12:38:13.568134Z","shell.execute_reply":"2023-04-19T12:38:13.625184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.628002Z","iopub.execute_input":"2023-04-19T12:38:13.628907Z","iopub.status.idle":"2023-04-19T12:38:13.661016Z","shell.execute_reply.started":"2023-04-19T12:38:13.628854Z","shell.execute_reply":"2023-04-19T12:38:13.659586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.664582Z","iopub.execute_input":"2023-04-19T12:38:13.664966Z","iopub.status.idle":"2023-04-19T12:38:13.691501Z","shell.execute_reply.started":"2023-04-19T12:38:13.66493Z","shell.execute_reply":"2023-04-19T12:38:13.69018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.69564Z","iopub.execute_input":"2023-04-19T12:38:13.696514Z","iopub.status.idle":"2023-04-19T12:38:13.715666Z","shell.execute_reply.started":"2023-04-19T12:38:13.696449Z","shell.execute_reply":"2023-04-19T12:38:13.714244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Audio(df_meta_data['filename'][0])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.722627Z","iopub.execute_input":"2023-04-19T12:38:13.723013Z","iopub.status.idle":"2023-04-19T12:38:13.753495Z","shell.execute_reply.started":"2023-04-19T12:38:13.722977Z","shell.execute_reply":"2023-04-19T12:38:13.752606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(df_meta_data['secondary_labels'] == \"[]\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.755027Z","iopub.execute_input":"2023-04-19T12:38:13.755368Z","iopub.status.idle":"2023-04-19T12:38:13.767845Z","shell.execute_reply.started":"2023-04-19T12:38:13.755334Z","shell.execute_reply":"2023-04-19T12:38:13.766524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**total of 16941 rows and 227 nulls in latitude and longitude**</br>\n**14636 nulls in the secondary label** </br>\n\n","metadata":{}},{"cell_type":"markdown","source":"## EDA columns","metadata":{}},{"cell_type":"markdown","source":"#### primary label: 264 distinct labels in which 52 labels has less than 11 samples and the max samples per label = 500 (in just 5 labels)","metadata":{}},{"cell_type":"code","source":"## primary label distribution ##\n\nprimary_label_counter = df_meta_data['primary_label'].value_counts().sort_values() ## this is a series with the index = primary class and value = count\nfig = px.bar(df_meta_data,  y = primary_label_counter.index , x = primary_label_counter.values,   )\nfig.update_layout(title_text=\"Distribution of Primary Labels\")\nfig.show()\nprint(primary_label_counter[-5:])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:13.769675Z","iopub.execute_input":"2023-04-19T12:38:13.770126Z","iopub.status.idle":"2023-04-19T12:38:15.442626Z","shell.execute_reply.started":"2023-04-19T12:38:13.770077Z","shell.execute_reply":"2023-04-19T12:38:15.441169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('number of classes that has only one row', (primary_label_counter == 1).sum() )\nprint('number of classes that has only two row', (primary_label_counter == 2).sum() )\nprint('number of classes that has only three row', (primary_label_counter == 3).sum() )\nprint('number of classes that has less than 5 row', (primary_label_counter <= 5).sum() )\nprint('number of classes that has less than 10 row', (primary_label_counter <= 10).sum() )\n\nprint()\nprint('number of classes that has less 500 row', (primary_label_counter == 500).sum() )\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.444263Z","iopub.execute_input":"2023-04-19T12:38:15.444739Z","iopub.status.idle":"2023-04-19T12:38:15.455044Z","shell.execute_reply.started":"2023-04-19T12:38:15.444699Z","shell.execute_reply":"2023-04-19T12:38:15.453484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"primary_label_counter [ primary_label_counter == 1 ] ## you have to make heavy augmentaiton for theses labels","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.457478Z","iopub.execute_input":"2023-04-19T12:38:15.458829Z","iopub.status.idle":"2023-04-19T12:38:15.479119Z","shell.execute_reply.started":"2023-04-19T12:38:15.458735Z","shell.execute_reply":"2023-04-19T12:38:15.477905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(primary_label_counter) )\nprint(max(primary_label_counter))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.480802Z","iopub.execute_input":"2023-04-19T12:38:15.481181Z","iopub.status.idle":"2023-04-19T12:38:15.488923Z","shell.execute_reply.started":"2023-04-19T12:38:15.481143Z","shell.execute_reply":"2023-04-19T12:38:15.487938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta_data['secondary_labels'] = df_meta_data['secondary_labels'].apply(eval)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.489864Z","iopub.execute_input":"2023-04-19T12:38:15.491058Z","iopub.status.idle":"2023-04-19T12:38:15.70066Z","shell.execute_reply.started":"2023-04-19T12:38:15.491018Z","shell.execute_reply":"2023-04-19T12:38:15.699196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"secondary_labels = list( itertools.chain.from_iterable(df_meta_data['secondary_labels']) )\nsecondary_labels_counter = Counter(secondary_labels)\nprint(secondary_labels_counter.most_common(5) )\n#px.bar()\nsecondary_labels_series = pd.Series(secondary_labels_counter ).sort_values(ascending=True)\nfig = px.bar(y = secondary_labels_series.index , x= secondary_labels_series.values)\nfig.update_layout(title_text = \"secondary labels distribution\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.702119Z","iopub.execute_input":"2023-04-19T12:38:15.702473Z","iopub.status.idle":"2023-04-19T12:38:15.770169Z","shell.execute_reply.started":"2023-04-19T12:38:15.702439Z","shell.execute_reply":"2023-04-19T12:38:15.768888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## type column","metadata":{}},{"cell_type":"code","source":"type_col = df_meta_data['type'].value_counts().sort_values()\nprint(len(type_col) ) ## 796 distinct values ## we don't know what type of calls are in the test dataset\npx.histogram(x = type_col.index , y = type_col.values)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.771678Z","iopub.execute_input":"2023-04-19T12:38:15.772273Z","iopub.status.idle":"2023-04-19T12:38:15.877274Z","shell.execute_reply.started":"2023-04-19T12:38:15.772235Z","shell.execute_reply":"2023-04-19T12:38:15.875993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### rating col","metadata":{}},{"cell_type":"code","source":"rating_col = df_meta_data['rating'].value_counts().sort_values(ascending=True)\nfig = px.bar(df_meta_data , x = rating_col.index , y=rating_col.values, )\nfig.update_layout(title_text = \"rating distribution\")\nfig.show()\nprint(rating_col.index)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.879003Z","iopub.execute_input":"2023-04-19T12:38:15.879488Z","iopub.status.idle":"2023-04-19T12:38:15.943176Z","shell.execute_reply.started":"2023-04-19T12:38:15.879438Z","shell.execute_reply":"2023-04-19T12:38:15.941769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.box(df_meta_data , y= 'rating')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:15.944771Z","iopub.execute_input":"2023-04-19T12:38:15.945143Z","iopub.status.idle":"2023-04-19T12:38:16.040243Z","shell.execute_reply.started":"2023-04-19T12:38:15.945106Z","shell.execute_reply":"2023-04-19T12:38:16.038964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### visualizing latitude and longitude on the map","metadata":{}},{"cell_type":"code","source":"fig = px.scatter_mapbox(df_meta_data,lat = 'latitude' , lon='longitude',color = 'primary_label' , zoom =3 , height = 500)\nfig.update_layout(title_text= \"distribution of locations on map\")\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.update_layout(margin={\"r\":0,\"t\":0,\"l\":0,\"b\":0})\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:16.042125Z","iopub.execute_input":"2023-04-19T12:38:16.042575Z","iopub.status.idle":"2023-04-19T12:38:16.839897Z","shell.execute_reply.started":"2023-04-19T12:38:16.042512Z","shell.execute_reply":"2023-04-19T12:38:16.838415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.density_mapbox(df_meta_data, lat='latitude', lon='longitude', radius=10,\n                        center=dict(lat=26, lon=30), zoom=4,\n                        mapbox_style=\"stamen-terrain\")\nfig.update_layout(title_text=\"Distribution of Bird Sightings\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:38:16.841433Z","iopub.execute_input":"2023-04-19T12:38:16.841844Z","iopub.status.idle":"2023-04-19T12:38:16.938483Z","shell.execute_reply.started":"2023-04-19T12:38:16.841803Z","shell.execute_reply":"2023-04-19T12:38:16.937167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### note that test data are from eastern african bird species ##","metadata":{}},{"cell_type":"markdown","source":"## bird species relationships","metadata":{}},{"cell_type":"code","source":"import pandas as pd\npath = \"/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv\"\ndf_species = pd.read_csv(path)\ndf_species.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T20:06:54.018035Z","iopub.execute_input":"2023-05-03T20:06:54.019378Z","iopub.status.idle":"2023-05-03T20:06:54.172421Z","shell.execute_reply.started":"2023-05-03T20:06:54.01933Z","shell.execute_reply":"2023-05-03T20:06:54.171154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar( df_species['FAMILY'].value_counts() )\nfig.show()\nprint(len( df_species['FAMILY'].unique()) )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T20:07:50.712644Z","iopub.execute_input":"2023-05-03T20:07:50.71327Z","iopub.status.idle":"2023-05-03T20:07:50.736563Z","shell.execute_reply.started":"2023-05-03T20:07:50.713214Z","shell.execute_reply":"2023-05-03T20:07:50.734396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"number of uniqe values\")\nfor col in df_species.columns:\n    print(col , len(df_species[col].unique()) )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T20:06:54.24443Z","iopub.status.idle":"2023-05-03T20:06:54.244906Z","shell.execute_reply.started":"2023-05-03T20:06:54.244658Z","shell.execute_reply":"2023-05-03T20:06:54.244682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_species['family']","metadata":{"execution":{"iopub.status.busy":"2023-05-03T20:06:54.247275Z","iopub.status.idle":"2023-05-03T20:06:54.247754Z","shell.execute_reply.started":"2023-05-03T20:06:54.247529Z","shell.execute_reply":"2023-05-03T20:06:54.247554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_species.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T12:41:52.324754Z","iopub.execute_input":"2023-04-19T12:41:52.325227Z","iopub.status.idle":"2023-04-19T12:41:52.346788Z","shell.execute_reply.started":"2023-04-19T12:41:52.325181Z","shell.execute_reply":"2023-04-19T12:41:52.345545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}