{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"#eda analysis on bird sounds dataset\n\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\nimport IPython.display as ipd\nimport librosa #module to analyse audio signals\n\nfrom bokeh.models import ColumnDataSource, LinearAxis, Range1d\nfrom bokeh.layouts import column, row\n\nfrom bokeh.models.tools import HoverTool\nfrom bokeh.palettes import BuGn4, PuRd, Reds, Set3\nfrom bokeh.plotting import figure, output_notebook, show\nfrom bokeh.transform import cumsum\n\noutput_notebook()\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = train = pd.read_csv('../input/birdsong-recognition/train.csv')\ntest = pd.read_csv('../input/birdsong-recognition/test.csv')\naudio_path = \"../input/birdsong-recognition/train_audio\"\n\ntrain_extend = pd.read_csv(\"../input/xeno-canto-bird-recordings-extended-a-m/train_extended.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.country\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[\"type\"].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_bird_map = train[['ebird_code', 'species']].drop_duplicates()\n\nfor ebird_code in os.listdir(audio_path)[:5]:\n    species = df_bird_map[df_bird_map['ebird_code'] == ebird_code].species.values[0]\n    audio_file = os.listdir(f\"{audio_path}/{ebird_code}\")[0]\n    path_to_audio = f\"{audio_path}/{ebird_code}/{audio_file}\"\n    ipd.display(ipd.HTML(f\"<h2>{ebird_code} ({species})</h2>\"))\n    ipd.display(ipd.Audio(path_to_audio))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#target variable is ebird code. After training, the audios should be classisfied \n#to the correct ebird label\n#looking at the distribution of ebird\n\nbird_type = train.groupby(\"ebird_code\")[\"filename\"].count().reset_index().rename(columns = {\"filename\": \"recordings\"}).sort_values(\"recordings\")\n\nsource = ColumnDataSource(bird_type)\ntooltips = [(\"Bird code\", \"@ebird_code\"),(\"Recordings\", \"@recordings\")]\n\nfig = figure(plot_width = 650, plot_height = 3000, y_range = bird_type['ebird_code'].values, tooltips = tooltips, title = \"Count of Bird Species\")\n\nfig.hbar(\"ebird_code\", right=\"recordings\", source = source, height = 0.75, color = \"blue\", alpha = 0.6)\n\nfig.xaxis.axis_label = \"Count\"\nfig.yaxis.axis_label = \"Ebird code\"\n\nshow(fig)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bird_type.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\"\"\"\nAbout half of the bird species have 100 plus recordings\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#looking at the times when the recordings were taken\n\ndf_date = train.groupby(\"date\")[\"ebird_code\"].count().reset_index().rename(columns = {'ebird_code': \"recordings\"})\ndf_date.date = pd.to_datetime(df_date.date, errors = 'coerce')\n#drop missing values \ndf_date.dropna(inplace = True)\n\ndf_date[\"weekday\"] = df_date[\"date\"].dt.day_name()\n\nsource_1 = ColumnDataSource(df_date)\n\ntooltips_1 = [(\"Date\", \"@date\"), (\"Recordings\", \"@recordings\")]\n\nformatters = {\"@date\": \"datetime\"}\n\nfig1 = figure(plot_width = 700, plot_height = 400, x_axis_type = \"datetime\", title = \"Date of recording\")\nfig1.line(\"date\", \"recordings\", source = source_1, width = 2, color = \"orange\", alpha =0.6)\n\nfig1.add_tools(HoverTool(tooltips = tooltips_1, formatters = formatters))\nfig1.xaxis.axis_label = \"Date\"\nfig1.yaxis.axis_label = \"Recordings\"\n\nshow(fig1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nMost of the recordings have been recorded since 2010\nThere is an uusual spike of recordings around the year 2003\n\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\n\n\n\ntrain[\"hour\"] = pd.to_numeric(train[\"time\"].str.split(\":\", expand = True)[0], errors = \"coerce\")\n\ndf_hour = train[train[\"hour\"].notna()].groupby(\"hour\")[\"ebird_code\"].count().reset_index().rename(columns ={\"ebird_code\" : \"recordings\"})\n\nsource_2 = ColumnDataSource(df_hour)\ntooltips_2 = [(\"Hour\", \"@hour\"),(\"Recordings\", \"@recordings\")]\n\nfig2 = figure( plot_width = 500, plot_height = 400, tooltips = tooltips_2, title = \"Hour of recordings\")\nfig2.vbar(\"hour\", top = \"recordings\", source = source_2, width = 0.75, color = \"tomato\", alpha = 0.6)\n\n\nfig2.xaxis.axis_label = \"Hour of day\"\nfig2.yaxis.axis_label = \"Recordings\"\n\nshow(fig2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nMost recordings happen from around 6am to 7pm\nlow number of recordings are during the night\nhigh number of recordings happen in between 6am and 12am\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_weekday = df_date.groupby(\"weekday\")[\"recordings\"].sum().reset_index().sort_values(\"recordings\", ascending=False)\n\nsource_3 = ColumnDataSource(df_weekday)\n\ntooltips_3 =[(\"Weekday\", \"@weekday\"),(\"Recordings\", \"@recordings\")]\n\nfig3 = figure(plot_width = 500, plot_height = 400, x_range = df_weekday[\"weekday\"].values, tooltips = tooltips_3, title = \"Day of the week and no of recordings\")\nfig3.vbar(\"weekday\", top=\"recordings\", source = source_3, width = 0.75, color=\"limegreen\", alpha=0.6 )\n\nfig3.xaxis.axis_label = \"Day of the week\"\nfig3.yaxis.axis_label = \"Recordings\"\n\nshow(fig3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nMost recordings happen during the weekends\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#countries with the highest number of recordings\n\ndf_country = train.groupby(\"country\")[\"ebird_code\"].count().reset_index().rename(columns = {\"ebird_code\": \"recordings\"}).sort_values(\"recordings\", ascending=False).head(20).sort_values(\"recordings\")\n\nsource_4 = ColumnDataSource(df_country)\n\ntooltips_4 = [(\"country\", '@country'), (\"recordings\", \"@recordings\")]\n\nfig4 = figure(plot_width = 650, plot_height = 700, y_range = df_country[\"country\"].values, tooltips = tooltips_1, title = \"Country of recording\")\nfig4.hbar(\"country\", right = \"recordings\", source = source_4, height = 0.75, color = \"coral\", alpha = 0.6)\n\n\nfig4.xaxis.axis_label = \"Country\"\nfig4.yaxis.axis_label = \"Recordings\"\nshow(fig4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#distribution of bird recordings by their location\ndf_location = train.groupby(\"location\")['ebird_code'].count().reset_index().rename(columns = {\"ebird_code\":\"recordings\"}).sort_values(\"recordings\", ascending = False).head(20).sort_values(\"recordings\")\n\nsource_5 = ColumnDataSource(df_location)\n\ntooltips_5 = [(\"Location\", \"@location\"), (\"Recoedings\", \"@recordings\")]\n\nfig5 = figure(plot_width = 650, plot_height = 700, y_range = df_location['location'].values, tooltips = tooltips_5, title = \"Top 20 locations for recordings\")\nfig5.hbar(\"location\", right = \"recordings\", source = source_5, height = 0.75, color = \"red\", alpha = 0.6)\nshow(fig5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#distribution of bird calls in training set\ndf_calltype = train.groupby(\"type\")[\"ebird_code\"].count().reset_index().rename(columns = {\"ebird_code\": \"records\"}).sort_values(\"records\", ascending = False).head(15)\n\nsource_6 = ColumnDataSource(df_calltype)\n\ntooltips_6 = [(\"type\", \"@type\"), (\"records\", \"@records\")]\n\nfig6 = figure(plot_width = 650, plot_height = 700, x_range = df_calltype['type'].values, tooltips = tooltips_6, title = \"Top song types\" )\nfig6.vbar(\"type\", top=\"records\", source = source_6, width = 0.75, color=\"purple\", alpha=0.6 )\n\nfig6.xaxis.axis_label = \"Bird song type\"\nfig6.yaxis.axis_label = \"Recordings\"\nshow(fig6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_extend.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_original = train.groupby(\"species\")[\"filename\"].count().reset_index().rename(columns = {\"filename\": \"recordings_original\"})\n\ndf_train_extend = train_extend.groupby(\"species\")[\"filename\"].count().reset_index().rename(columns = {\"filename\": \"recordings_extended\"})\n\ndf_bird = df_train_original.merge(df_train_extend, on = \"species\", how = \"left\").fillna(0)\ndf_bird[\"recordings_total\"] = df_bird[\"recordings_original\"] + df_bird[\"recordings_extended\"]\n\ndf_bird = df_bird.sort_values(\"recordings_total\").reset_index()\n\nsource_6 = ColumnDataSource(df_bird)\ntooltips_6 = [(\"Species\", \"@species\"), (\"original recordings\", \"recordings_original\"), (\"extended recordings\", \"@recordings_extended\")]\n\nfig6 = figure(plot_width = 750, plot_height = 3000, y_range = df_bird.species.values, tooltips = tooltips_6, title = \"Count of Bird Species\")\nfig6.hbar_stack([\"recordings_original\", \"recordings_extended\"], y = \"species\", source = source_6, height = 0.75, color = [\"blue\", \"orange\"], alpha = 0.65)\n              \nfig6.xaxis.axis_label = \"Count\"\nfig6.yaxis.axis_label = \"Species\"\n\nshow(fig6)\n              ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_audio_path = \"../input/birdsong-recognition/train_audio/aldfly/XC135455.mp3\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import librosa.display","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#analysing the audio\nx , sr = librosa.load(sample_audio_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(x.shape, sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#visualising audio\nplt.figure(figsize = (14,5))\nlibrosa.display.waveplot(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}