{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"# 1. Introduction\n\n\nIn this competition the researchers from Cornell Lab of Ornithology’s Center for Conservation Bioacoustics (CBC) wants the Kaggle community to help them build an AI solution to identify bird species using their bird call audio.\n\n<img src=\"https://images.unsplash.com/photo-1493236296276-d17357e28888?ixlib=rb-1.2.1&ixid=eyJhcHBfaWQiOjEyMDd9&auto=format&fit=crop&w=1051&q=80\" width=\"800\"></img>","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# 2. Analysis preparation\n\n## 2.1. Load packages\n\nHere we load the Python modules we will need for our analysis.","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook\n%matplotlib inline \nimport IPython as ipy\nimport IPython.display as ipyd\nimport librosa\nimport librosa.display\nimport folium\nfrom folium.plugins import HeatMap, HeatMapWithTime\nimport plotly.express as px\nimport sklearn\nimport warnings\nwarnings.filterwarnings(action='ignore')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"## 2.2. Load the data\n\nHere we load the metadata (csv file).","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/birdsong-recognition/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"\n## 2.3. Glimpse the data\n\nWe perform a preliminary analysis of the data, looking to such things like data shape, missing data, unique values.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"train data: {train_df.shape}\")\nprint(f\"train data columns: {list(train_df.columns)}\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"pd.set_option('display.max_columns', 50)\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"train_df.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 2.4 Missing data","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum()\n    percent = (data.isnull().sum()/data.isnull().count()*100)\n    tt = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    types = []\n    for col in data.columns:\n        dtype = str(data[col].dtype)\n        types.append(dtype)\n    tt['Types'] = types\n    return(np.transpose(tt))","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"missing_data(train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 2.5. Unique values","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return(np.transpose(tt))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unique_values(train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3. Data exploration\n\nWe will explore the data, starting with the metadata information (csv file).\n\n## 3.1. Features values distribution","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n    '''\n    Plot count of classes / feature\n    param: feature - the feature to analyze\n    param: title - title to add to the graph\n    param: df - dataframe from which we plot feature's classes distribution \n    param: size - default 1.\n    '''\n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    g = sns.countplot(df[feature], order = df[feature].value_counts().index[:20], palette='Set1')\n    g.set_title(\"Number and percentage of {}\".format(title))\n    if(size > 2):\n        plt.xticks(rotation=90, size=8)\n    for p in ax.patches:\n        height = p.get_height()\n        ax.text(p.get_x()+p.get_width()/2.,\n                height + 3,\n                '{:1.2f}%'.format(100*height/total),\n                ha=\"center\") \n    plt.show()  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"playback_used values: {train_df.playback_used.nunique()}\")\nplot_count(\"playback_used\", \"playback_used\", train_df, size=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"ebird_codes values: {train_df.ebird_code.nunique()}\")\nplot_count(\"ebird_code\", \"ebird_code (first 20 entries)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"channels values: {train_df.channels.nunique()}\")\nplot_count(\"channels\", \"channels\", train_df, size=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"pitch values: {train_df.pitch.nunique()}\")\nplot_count(\"pitch\", \"pitch\", train_df, size=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"speed values: {train_df.speed.nunique()}\")\nplot_count(\"speed\", \"speed\", train_df, size=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"species values: {train_df.species.nunique()}\")\nplot_count(\"species\", \"species (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"E-bird codes and Species seems to be corresponding values.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"number of notes values: {train_df.number_of_notes.nunique()}\")\nplot_count(\"number_of_notes\", \"number_of_notes\", train_df, size=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"bird_seen values: {train_df.bird_seen.nunique()}\")\nplot_count(\"bird_seen\", \"bird_seen\", train_df, size=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"sci_name values: {train_df.sci_name.nunique()}\")\nplot_count(\"sci_name\", \"sci_name (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"location values: {train_df.location.nunique()}\")\nplot_count(\"location\", \"location (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"sampling_rate values: {train_df.sampling_rate.nunique()}\")\nplot_count(\"sampling_rate\", \"sampling_rate\", train_df, size=3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"type values: {train_df.type.nunique()}\")\nplot_count(\"type\", \"type (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"elevation values: {train_df.elevation.nunique()}\")\nplot_count(\"elevation\", \"elevation (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"latitude values: {train_df.latitude.nunique()}\")\nplot_count(\"latitude\", \"latitude (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"longitude values: {train_df.longitude.nunique()}\")\nplot_count(\"longitude\", \"longitude (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":false},"cell_type":"markdown","source":"Latitude, longitude, elevation can be used to build a map with the observation location and altitude.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"bitrate_of_mp3 values: {train_df.bitrate_of_mp3.nunique()}\")\nplot_count(\"bitrate_of_mp3\", \"bitrate_of_mp3 (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"volume values: {train_df.volume.nunique()}\")\nplot_count(\"volume\", \"volume\", train_df, size=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"file_type values: {train_df.file_type.nunique()}\")\nplot_count(\"file_type\", \"file_type\", train_df, size=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"background values: {train_df.background.nunique()}\")\nplot_count(\"background\", \"background (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Background is given in name of the species and (in paranthesys) the scientific name.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"author values: {train_df.author.nunique()}\")\nplot_count(\"author\", \"author (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"primary_label values: {train_df.primary_label.nunique()}\")\nplot_count(\"primary_label\", \"primary_label (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"length values: {train_df.length.nunique()}\")\nplot_count(\"length\", \"length\", train_df, size=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"time values: {train_df.time.nunique()}\")\nplot_count(\"time\", \"time (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"country values: {train_df.country.nunique()}\")\nplot_count(\"country\", \"country (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"recordist values: {train_df.recordist.nunique()}\")\nplot_count(\"recordist\", \"recordist (first 20)\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print(f\"license values: {train_df.license.nunique()}\")\nplot_count(\"license\", \"license\", train_df, size=3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 3.2. Geographical distribution\n\nLet's look now to the geographical distribution of data. We will group on latitude and longitude and count the occurences for each {latitude, longitude} tuple.\nNext, we will represent this geographical distribution with a heatmap, the intensity of color being proportional with the number of data.","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"tmp = train_df.groupby(['latitude', 'longitude'])['url'].count()\nlatlong_df = pd.DataFrame(tmp).reset_index()\nlatlong_df.columns = ['latitude', 'longitude', 'count']\nlatlong_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"latlong_df = latlong_df.loc[~(latlong_df.latitude==\"Not specified\")]","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"m = folium.Map(location=[0,0], zoom_start=2)\nmax_val = max(latlong_df['count'])\nHeatMap(data=latlong_df[['latitude', 'longitude', 'count']],\\\n        radius=15, max_zoom=12).add_to(m)\nm","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We also can group the data on countries.","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"tmp = train_df.groupby(['country'])['url'].count()\ncountry_df = pd.DataFrame(tmp).reset_index()\ncountry_df.columns = ['country','count']\ndf = px.data.gapminder().query(\"year==2007\")\ndf = df[['country', 'iso_alpha']]\ncountry_df = country_df.merge(df, on=\"country\")\ncountry_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"hover_text = []\nfor index, row in country_df.iterrows():\n    hover_text.append((f\"country: {row['country']}<br>count: {row['count']}<br>country code: {row['iso_alpha']}\"))\ncountry_df['hover_text'] = hover_text\n\nfig = px.choropleth(country_df, \n                    locations=\"iso_alpha\",\n                    hover_name='hover_text',\n                    color=\"count\",\n                     projection=\"natural earth\",\n                    color_continuous_scale=px.colors.sequential.Plasma,\n                    width=700, height=525)\nfig.update_geos(   \n    showcoastlines=True, coastlinecolor=\"DarkBlue\",\n    showland=True, landcolor=\"LightGrey\",\n    showocean=True, oceancolor=\"LightBlue\",\n    showlakes=True, lakecolor=\"Blue\",\n    showrivers=True, rivercolor=\"Blue\",\n    showcountries=True, countrycolor=\"DarkBlue\"\n)\nfig.update_layout(title = 'Number of observations per country<br>(hover for details)')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## 3.3. Time and location distribution","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"train_df['dated'] = pd.to_datetime(train_df['date'], format='%Y-%m-%d', errors='coerce')","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"train_df['year'] = train_df['dated'].dt.year\ntrain_df['month'] = train_df['dated'].dt.month\ntrain_df['day'] = train_df['dated'].dt.day\ntrain_df['dayofweek'] = train_df['dated'].dt.dayofweek","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def plot_time_variation(df, x='date', y='count', hue=None, size=1, is_log=False):\n    f, ax = plt.subplots(1,1, figsize=(4*size,3*size))\n    g = sns.lineplot(x=x, y=y, hue=hue, data=df)\n    plt.xticks(rotation=90)\n    if hue:\n        plt.title(f'{y} grouped by {hue}')\n    else:\n        plt.title(f'{y}')\n    if(is_log):\n        ax.set(yscale=\"log\")\n    ax.grid(color='black', linestyle='dotted', linewidth=0.75)\n    plt.show() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_df = train_df.groupby(['year'])['url'].count().reset_index()\nagg_df.columns = ['year', 'count']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_time_variation(agg_df, x='year', y=\"count\", hue=None, size=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.columns","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"agg_df = train_df.groupby(['year', 'bird_seen'])['url'].count().reset_index()\nagg_df.columns = ['year', 'bird_seen', 'count']\nplot_time_variation(agg_df, x='year', y=\"count\", hue='bird_seen', size=4, is_log=True)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"agg_df = train_df.groupby(['year', 'playback_used'])['url'].count().reset_index()\nagg_df.columns = ['year', 'playback_used', 'count']\nplot_time_variation(agg_df, x='year', y=\"count\", hue='playback_used', size=4, is_log=True)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"agg_df = train_df.groupby(['year', 'license'])['url'].count().reset_index()\nagg_df.columns = ['year', 'license', 'count']\nplot_time_variation(agg_df, x='year', y=\"count\", hue='license', size=4, is_log=True)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"print(f\"year values: {train_df.year.nunique()}\")\nplot_count(\"year\", \"year\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"print(f\"month values: {train_df.month.nunique()}\")\nplot_count(\"month\", \"month\", train_df, size=3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Going out to record birdsongs happens mostly in May and Junr, when more than 40% of all records were made.","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"print(f\"day values: {train_df.day.nunique()}\")\nplot_count(\"day\", \"day\", train_df, size=4)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"print(f\"dayofweek values: {train_df.dayofweek.nunique()}\")\nplot_count(\"dayofweek\", \"dayofweek\", train_df, size=3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It looks like recording birdsongs is mainly a weekend activity (which makes sense, since most of the recorders are volunteers), since most of the recording are on Saturdays & Sundays.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Signal data exploration","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Let's explore now the signal data from the training set.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"TRAIN_AUDIO_PATH = \"../input/birdsong-recognition/train_audio/\"\nfiles = os.listdir(TRAIN_AUDIO_PATH)\nprint(f\"train folders: {len(files)}\")\nprint(f\"some ebird_code examples: {files[0:10]}\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Play audio\n\nLet's listen to some of the audio signals.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def play_audio_file(ebird_code, samples=3):\n    for sample in range(0, samples):\n        file_name = train_df.loc[train_df.ebird_code==ebird_code, \"filename\"].values[sample]\n        length = train_df.loc[train_df.ebird_code==ebird_code, \"length\"].values[sample]\n        file_type = train_df.loc[train_df.ebird_code==ebird_code, \"file_type\"].values[sample]\n        volume = train_df.loc[train_df.ebird_code==ebird_code, \"volume\"].values[sample]\n        bitrate_of_mp3 = train_df.loc[train_df.ebird_code==ebird_code, \"bitrate_of_mp3\"].values[sample]\n        audio_file_path = os.path.join(TRAIN_AUDIO_PATH, ebird_code, file_name)\n        print(f\"ebird_code: {ebird_code} file: {file_name}\\nlength: {length}\\nvolume: {volume}\\nbit rate: {bitrate_of_mp3}\\nfile type: {file_type}\")\n        ipy.display.display(ipyd.Audio(audio_file_path))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"play_audio_file(\"aldfly\", 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"play_audio_file(\"purfin\", 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"play_audio_file(\"marwre\", 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"play_audio_file(\"boboli\", 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"play_audio_file(\"wewpew\", 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"play_audio_file(\"eawpew\", 2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Analyze signals\n\n\n### Signal plots\n\nLet's plot some of the signals in time.\n\nWe create a function that samples few signals from a certain species and display it.","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def plot_audio_file(ebird_code):\n\n    plt.figure(figsize=(16,6))\n    sample = 0\n    file_name = train_df.loc[train_df.ebird_code==ebird_code, \"filename\"].values[sample]\n    length = train_df.loc[train_df.ebird_code==ebird_code, \"length\"].values[sample]\n    file_type = train_df.loc[train_df.ebird_code==ebird_code, \"file_type\"].values[sample]\n    volume = train_df.loc[train_df.ebird_code==ebird_code, \"volume\"].values[sample]\n    bitrate_of_mp3 = train_df.loc[train_df.ebird_code==ebird_code, \"bitrate_of_mp3\"].values[sample]\n    audio_file_path = os.path.join(TRAIN_AUDIO_PATH, ebird_code, file_name)\n    x , sr = librosa.load(audio_file_path)\n    librosa.display.waveplot(x, sr=sr)\n    plt.gca().set_title(f\"ebird_code: {ebird_code} file: {file_name}\\nlength: {length} volume: {volume} bit rate: {bitrate_of_mp3} file type: {file_type}\")\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"plot_audio_file(\"aldfly\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"plot_audio_file(\"purfin\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"plot_audio_file(\"marwre\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"plot_audio_file(\"brebla\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file(\"boboli\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file(\"wewpew\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file(\"eawpew\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Signal spectrogram\n\nLet's also plot some signal spectrograms. A spectrogram is a visual representation of the spectre of frequencies associated with a signal.\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_audio_file_spectrogram(ebird_code):\n\n    plt.figure(figsize=(16,6))\n    sample = 0\n    file_name = train_df.loc[train_df.ebird_code==ebird_code, \"filename\"].values[sample]\n    length = train_df.loc[train_df.ebird_code==ebird_code, \"length\"].values[sample]\n    file_type = train_df.loc[train_df.ebird_code==ebird_code, \"file_type\"].values[sample]\n    volume = train_df.loc[train_df.ebird_code==ebird_code, \"volume\"].values[sample]\n    bitrate_of_mp3 = train_df.loc[train_df.ebird_code==ebird_code, \"bitrate_of_mp3\"].values[sample]\n    audio_file_path = os.path.join(TRAIN_AUDIO_PATH, ebird_code, file_name)\n    x , sr = librosa.load(audio_file_path)\n    xs = librosa.stft(x)\n    xdb = librosa.amplitude_to_db(abs(xs))\n    librosa.display.specshow(xdb, sr=sr, x_axis='time', y_axis='hz')\n    plt.gca().set_title(f\"Spectrogram - ebird_code: {ebird_code} file: {file_name}\\nlength: {length} volume: {volume} bit rate: {bitrate_of_mp3} file type: {file_type}\")\n    plt.colorbar()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"aldfly\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"purfin\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"marwre\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"brebla\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"boboli\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"wewpew\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectrogram(\"eawpew\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Signal spectral rolloff\n\nThe spectral rolloff is a measure of the shape of the signal, representing the frequency at which high frequencies decline to 0. Can be calculated by the fraction of bins in the power spectrum where 85% of its power is at lower frequencies.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def normalize(x, axis=0):\n    return sklearn.preprocessing.minmax_scale(x, axis=axis)\n\ndef plot_audio_file_spectral_rolloff(ebird_code):\n\n    plt.figure(figsize=(16,6))\n    sample = 0\n    file_name = train_df.loc[train_df.ebird_code==ebird_code, \"filename\"].values[sample]\n    length = train_df.loc[train_df.ebird_code==ebird_code, \"length\"].values[sample]\n    file_type = train_df.loc[train_df.ebird_code==ebird_code, \"file_type\"].values[sample]\n    volume = train_df.loc[train_df.ebird_code==ebird_code, \"volume\"].values[sample]\n    bitrate_of_mp3 = train_df.loc[train_df.ebird_code==ebird_code, \"bitrate_of_mp3\"].values[sample]\n    audio_file_path = os.path.join(TRAIN_AUDIO_PATH, ebird_code, file_name)\n    x , sr = librosa.load(audio_file_path)\n    spectral_rolloff = librosa.feature.spectral_rolloff(x+0.01, sr=sr)[0]\n    spectral_centroids = librosa.feature.spectral_centroid(x, sr=sr)[0]\n    frames = range(len(spectral_centroids))\n    t = librosa.frames_to_time(frames)\n    librosa.display.waveplot(x, sr=sr, alpha=0.4)\n    plt.gca().set_title(f\"Spectral rolloff - ebird_code: {ebird_code} file: {file_name}\\nlength: {length} volume: {volume} bit rate: {bitrate_of_mp3} file type: {file_type}\")\n    plt.plot(t, normalize(spectral_rolloff), color='r')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"aldfly\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"purfin\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"marwre\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"brebla\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"boboli\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"wewpew\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_audio_file_spectral_rolloff(\"eawpew\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}