{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\"\"\"\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\"\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n%matplotlib inline\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.graph_objs as go\n\n# Import data\ntrain_csv = pd.read_csv(\"../input/birdsong-recognition/train.csv\")\ntrain_csv.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1. Number of bird species","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"There are {:,} unique bird species in the dataset.\".format(len(train_csv['species'].unique())))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. Number of audio files","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"There are {:,} audio files in the dataset.\".format(len(train_csv['filename'].unique())))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3. Number of individuals for each species","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot the counts of first 20 bird species sorted by quantity\ntrain_csv['species'].value_counts().head(20).plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot the counts of first 20 bird species sorted alphabetically\ntrain_csv['species'].value_counts().sort_index(ascending=True).head(20).plot.bar()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 4. Elevation","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"top_10 = list(train_csv['elevation'].value_counts().head(10).reset_index()['index'])\ndata = train_csv[train_csv['elevation'].isin(top_10)]\n\nplt.figure(figsize=(16, 6))\nax = sns.countplot(data['elevation'], palette=\"hls\", order = data['elevation'].value_counts().index)\n\nplt.title(\"Top 10 Elevation Types\", fontsize=16)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 5. Countries","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"top_10 = list(train_csv['country'].value_counts().head(10).reset_index()['index'])\ndata = train_csv[train_csv['country'].isin(top_10)]\n\nplt.figure(figsize=(16, 6))\nax = sns.countplot(data['country'], palette=\"hls\", order = data['country'].value_counts().index)\n\nplt.title(\"Top 10 Countries with bird recordings\", fontsize=16)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 6. Dates of recording","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_year(date):\n    return date.split('-')[0]\n\ntrain_csv['year'] = train_csv['date'].apply(get_year)\n\ntop_25 = list(train_csv['year'].value_counts().head(25).reset_index()['index'])\ndata = train_csv[train_csv['year'].isin(top_25)]\n\nplt.figure(figsize=(16, 6))\nax = sns.countplot(data['year'], palette=\"hls\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 7. Bird seen","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv['bird_seen'].fillna('Not Defined',inplace=True)\nlabels = train_csv['bird_seen'].value_counts().index\nvalues = train_csv['bird_seen'].value_counts().values\ncolors=['#3795bf','#bfbfbf', '#cf5353']\n\nfig = go.Figure(data=[go.Pie(labels=labels, values=values, textinfo='label+percent',\n                             insidetextorientation='radial',marker=dict(colors=colors))])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 8. Number of notes","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nax = sns.countplot(data['number_of_notes'], palette=\"hls\", order = data['number_of_notes'].value_counts().index)\nplt.xlabel(\"Number of notes\", fontsize=14)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 9. Playback Used","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv['playback_used'].fillna('Not Defined',inplace=True)\nlabels = train_csv['playback_used'].value_counts().index\nvalues = train_csv['playback_used'].value_counts().values\ncolors=['#3795bf','#bfbfbf', '#cf5353']\n\nfig = go.Figure(data=[go.Pie(labels=labels, values=values, textinfo='label+percent',\n                             insidetextorientation='radial',marker=dict(colors=colors))])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 10. Duration","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_csv['duration'].describe())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 11. Channels","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv['channels'].fillna('Not Defined',inplace=True)\nlabels = train_csv['channels'].value_counts().index\nvalues = train_csv['channels'].value_counts().values\ncolors=['#3795bf','#bfbfbf', '#cf5353']\n\nfig = go.Figure(data=[go.Pie(labels=labels, values=values, textinfo='label+percent',\n                             insidetextorientation='radial',marker=dict(colors=colors))])\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Reference:<br/>\nhttps://towardsdatascience.com/15-data-exploration-techniques-to-go-from-data-to-insights-93f66e6805df <br/>\nhttps://www.kaggle.com/andradaolteanu/birdcall-recognition-eda-and-audio-fe <br/>\nhttps://www.kaggle.com/parulpandey/eda-and-audio-processing-with-python","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}