{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-03T17:36:43.679834Z","iopub.execute_input":"2021-06-03T17:36:43.680413Z","iopub.status.idle":"2021-06-03T17:37:07.656132Z","shell.execute_reply.started":"2021-06-03T17:36:43.680274Z","shell.execute_reply":"2021-06-03T17:37:07.654949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport soundfile as sf\nimport librosa\nimport librosa.display\nimport IPython.display as display\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nimport librosa\nimport librosa.display\nimport os\nfrom tqdm import tqdm\nimport sklearn\nimport seaborn as sns\nimport plotly.express as px\n\n\nimport geopandas as gpd\nfrom shapely.geometry import Point, Polygon\n\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.utils import Sequence\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv1D, MaxPool1D, BatchNormalization\nfrom keras.optimizers import RMSprop,Adam\nfrom keras.applications import VGG19, VGG16, ResNet50\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:38:56.470661Z","iopub.execute_input":"2021-06-03T17:38:56.471103Z","iopub.status.idle":"2021-06-03T17:39:07.791292Z","shell.execute_reply.started":"2021-06-03T17:38:56.471068Z","shell.execute_reply":"2021-06-03T17:39:07.790050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/birdclef-2021/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:12.604840Z","iopub.execute_input":"2021-06-03T17:39:12.605240Z","iopub.status.idle":"2021-06-03T17:39:12.616574Z","shell.execute_reply.started":"2021-06-03T17:39:12.605206Z","shell.execute_reply":"2021-06-03T17:39:12.615184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_ogg_file(path, file):\n    \"\"\" Read audio files and returning the numpay array and the samplerate\"\"\"\n    \n    data, samplerate = sf.read(path+file)\n    return data, samplerate\n\n\ndef plot_audio_file(data, samplerate):\n    \"\"\" Plot the audio data\"\"\"\n    \n    sr = samplerate\n    fig = plt.figure(figsize=(8, 4))\n    x = range(len(data))\n    y = data\n    plt.plot(x, y)\n    plt.plot(x, y, color='red')\n    plt.legend(loc='upper center')\n    plt.grid()\n    ","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:15.179513Z","iopub.execute_input":"2021-06-03T17:39:15.179943Z","iopub.status.idle":"2021-06-03T17:39:15.188588Z","shell.execute_reply.started":"2021-06-03T17:39:15.179904Z","shell.execute_reply":"2021-06-03T17:39:15.187119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot spectrogram with mel scaling\ndef plot_spectrogram(data, samplerate):\n    sr = samplerate\n    spectrogram = librosa.feature.melspectrogram(data, sr=sr)\n    log_spectrogram = librosa.power_to_db(spectrogram, ref=np.max)\n    librosa.display.specshow(log_spectrogram, sr=sr, x_axis='time', y_axis='mel')","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:27.455201Z","iopub.execute_input":"2021-06-03T17:39:27.455605Z","iopub.status.idle":"2021-06-03T17:39:27.461946Z","shell.execute_reply.started":"2021-06-03T17:39:27.455562Z","shell.execute_reply":"2021-06-03T17:39:27.460848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(path+'train_soundscape_labels.csv')\ntrain_meta = pd.read_csv(path+'train_metadata.csv')\ntest_data = pd.read_csv(path+'test.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:30.094548Z","iopub.execute_input":"2021-06-03T17:39:30.094962Z","iopub.status.idle":"2021-06-03T17:39:30.633930Z","shell.execute_reply.started":"2021-06-03T17:39:30.094928Z","shell.execute_reply":"2021-06-03T17:39:30.633117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number train label samples:', len(train_labels))\nprint('Number train meta samples:', len(train_meta))\nprint('Number train short folder:', len(os.listdir(path+'train_short_audio')))\nprint('Number train audios:', len(os.listdir(path+'train_soundscapes')))\nprint('Number test samples:', len(test_data))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:32.804654Z","iopub.execute_input":"2021-06-03T17:39:32.805206Z","iopub.status.idle":"2021-06-03T17:39:32.818358Z","shell.execute_reply.started":"2021-06-03T17:39:32.805168Z","shell.execute_reply":"2021-06-03T17:39:32.816699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(path+'train_short_audio/caltow')[:2]","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:34.299606Z","iopub.execute_input":"2021-06-03T17:39:34.300043Z","iopub.status.idle":"2021-06-03T17:39:34.309883Z","shell.execute_reply.started":"2021-06-03T17:39:34.300001Z","shell.execute_reply":"2021-06-03T17:39:34.308650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:36.054596Z","iopub.execute_input":"2021-06-03T17:39:36.055055Z","iopub.status.idle":"2021-06-03T17:39:36.084242Z","shell.execute_reply.started":"2021-06-03T17:39:36.055013Z","shell.execute_reply":"2021-06-03T17:39:36.082857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:38.389597Z","iopub.execute_input":"2021-06-03T17:39:38.390052Z","iopub.status.idle":"2021-06-03T17:39:38.414658Z","shell.execute_reply.started":"2021-06-03T17:39:38.390011Z","shell.execute_reply":"2021-06-03T17:39:38.413358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DATA SIZE\n\nprint(f\"Training Dataset Shape: {(train_meta.shape)}\")\nprint(f\"Training Dataset Labels Shape: {(train_labels.shape)}\")","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:40.134707Z","iopub.execute_input":"2021-06-03T17:39:40.135168Z","iopub.status.idle":"2021-06-03T17:39:40.141427Z","shell.execute_reply.started":"2021-06-03T17:39:40.135130Z","shell.execute_reply":"2021-06-03T17:39:40.140341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique values in each column\n\nprint(\"Data: train\")\nfor col in train_meta.columns:\n    print(col + \":\" + (str(len(train_meta[col].unique()))))\n\nprint(\"\\nData: train_labels\")\nfor col in train_labels.columns:\n    print(col + \":\" + (str(len(train_labels[col].unique()))))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:42.654524Z","iopub.execute_input":"2021-06-03T17:39:42.655034Z","iopub.status.idle":"2021-06-03T17:39:42.772841Z","shell.execute_reply.started":"2021-06-03T17:39:42.654987Z","shell.execute_reply":"2021-06-03T17:39:42.771816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Time of the various recording\ntrain_meta['year'] = train_meta['date'].apply(lambda x: x.split(\"-\")[0])\ntrain_meta['month'] = train_meta['date'].apply(lambda x: x.split(\"-\")[1])\nplt.figure(figsize=(16, 6))\nax = sns.countplot(train_meta['year'].sort_values(ascending=False), palette=\"hls\")\n\nplt.title(\"Audio Files Registration per Year Made\", fontsize=16)\nplt.xticks(rotation=90, fontsize=13)\nplt.yticks(fontsize=13)\nplt.ylabel(\"Frequency\", fontsize=14)\nplt.xlabel(\"\");","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:44.254522Z","iopub.execute_input":"2021-06-03T17:39:44.254931Z","iopub.status.idle":"2021-06-03T17:39:45.350633Z","shell.execute_reply.started":"2021-06-03T17:39:44.254894Z","shell.execute_reply":"2021-06-03T17:39:45.349571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 6))\nax = sns.countplot(train_meta['month'].sort_values(ascending=False), palette=\"hls\")\n\nplt.title(\"Audio Files Registration per Month Made\", fontsize=16)\nplt.xticks(fontsize=13)\nplt.yticks(fontsize=13)\nplt.ylabel(\"Frequency\", fontsize=14)\nplt.xlabel(\"\");","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:48.155453Z","iopub.execute_input":"2021-06-03T17:39:48.156084Z","iopub.status.idle":"2021-06-03T17:39:48.744368Z","shell.execute_reply.started":"2021-06-03T17:39:48.156030Z","shell.execute_reply":"2021-06-03T17:39:48.743634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = 0\ntrain_meta.iloc[row]","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:50.344710Z","iopub.execute_input":"2021-06-03T17:39:50.345203Z","iopub.status.idle":"2021-06-03T17:39:50.353696Z","shell.execute_reply.started":"2021-06-03T17:39:50.345160Z","shell.execute_reply":"2021-06-03T17:39:50.352836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = train_meta.loc[row, 'primary_label']\nfilename = train_meta.loc[row, 'filename']\n\n# Check if the file is in the folder\nfilename in os.listdir(path+'train_short_audio/'+label)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:52.604944Z","iopub.execute_input":"2021-06-03T17:39:52.605653Z","iopub.status.idle":"2021-06-03T17:39:52.620419Z","shell.execute_reply.started":"2021-06-03T17:39:52.605609Z","shell.execute_reply":"2021-06-03T17:39:52.619012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename = '../input/birdclef-2021/train_short_audio/acafly/XC11209.ogg'","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:54.489638Z","iopub.execute_input":"2021-06-03T17:39:54.490029Z","iopub.status.idle":"2021-06-03T17:39:54.494567Z","shell.execute_reply.started":"2021-06-03T17:39:54.489995Z","shell.execute_reply":"2021-06-03T17:39:54.493662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n#default sr=22050\n# input is converted to mono by default\ndata, sample_rate = librosa.load(filename)\nlibrosa.display.waveplot(data, sr=sample_rate)\nprint(\"Sample Rate: \", sample_rate)\nipd.Audio(filename)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:56.394656Z","iopub.execute_input":"2021-06-03T17:39:56.395231Z","iopub.status.idle":"2021-06-03T17:39:57.820109Z","shell.execute_reply.started":"2021-06-03T17:39:56.395194Z","shell.execute_reply":"2021-06-03T17:39:57.818817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#spectrogram\n\nplt.figure(figsize=(18, 5))\nX = librosa.stft(data)\nXdb = librosa.amplitude_to_db(abs(X))\nlibrosa.display.specshow(Xdb, sr=sample_rate, x_axis='time', y_axis='hz')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:57.854133Z","iopub.execute_input":"2021-06-03T17:39:57.854494Z","iopub.status.idle":"2021-06-03T17:39:58.597828Z","shell.execute_reply.started":"2021-06-03T17:39:57.854457Z","shell.execute_reply":"2021-06-03T17:39:58.596450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#spectral centroid\n\nspectral_centroids = librosa.feature.spectral_centroid(data, sr=sample_rate)[0]\nplt.figure(figsize=(25, 9))\nframes = range(len(spectral_centroids))\nt = librosa.frames_to_time(frames)\n\n# Normalising the spectral centroid for visualisation\ndef normalize(x, axis=0):\n    return sklearn.preprocessing.minmax_scale(x, axis=axis)\n\n#Plotting the Spectral Centroid along with the waveform\nlibrosa.display.waveplot(data, sr=sample_rate, alpha=0.4)\nplt.plot(t, normalize(spectral_centroids), color='b')","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:58.789347Z","iopub.execute_input":"2021-06-03T17:39:58.789703Z","iopub.status.idle":"2021-06-03T17:39:59.277523Z","shell.execute_reply.started":"2021-06-03T17:39:58.789674Z","shell.execute_reply":"2021-06-03T17:39:59.276324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#spectral roll off\n\nplt.figure(figsize=(25, 9))\nspectral_rolloff = librosa.feature.spectral_rolloff(data+0.01, sr=sample_rate)[0]\nlibrosa.display.waveplot(data, sr=sample_rate, alpha=0.4)\nplt.plot(t, normalize(spectral_rolloff), color='r')","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:39:59.614727Z","iopub.execute_input":"2021-06-03T17:39:59.615181Z","iopub.status.idle":"2021-06-03T17:40:00.116887Z","shell.execute_reply.started":"2021-06-03T17:39:59.615145Z","shell.execute_reply":"2021-06-03T17:40:00.115515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#spectral bandwidth\nspectral_bandwidth_2 = librosa.feature.spectral_bandwidth(data+0.01, sr=sample_rate)[0]\nspectral_bandwidth_3 = librosa.feature.spectral_bandwidth(data+0.01, sr=sample_rate, p=3)[0]\nspectral_bandwidth_4 = librosa.feature.spectral_bandwidth(data+0.01, sr=sample_rate, p=4)[0]\nplt.figure(figsize=(25, 9))\nlibrosa.display.waveplot(data, sr=sample_rate, alpha=0.4)\nplt.plot(t, normalize(spectral_bandwidth_2), color='r')\nplt.plot(t, normalize(spectral_bandwidth_3), color='g')\nplt.plot(t, normalize(spectral_bandwidth_4), color='y')\nplt.legend(('p = 2', 'p = 3', 'p = 4'))  # p: order of spectral bandwidth","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:00.574999Z","iopub.execute_input":"2021-06-03T17:40:00.575463Z","iopub.status.idle":"2021-06-03T17:40:01.382152Z","shell.execute_reply.started":"2021-06-03T17:40:00.575417Z","shell.execute_reply":"2021-06-03T17:40:01.380713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ZCR\n\nplt.figure(figsize=(25, 9))\n# librosa.display.waveplot(data, sr=sample_rate)\n# Zooming in\nn0 = 9000\nn1 = 9100\n\nplt.plot(data[n0:n1])\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:01.844658Z","iopub.execute_input":"2021-06-03T17:40:01.845094Z","iopub.status.idle":"2021-06-03T17:40:02.059329Z","shell.execute_reply.started":"2021-06-03T17:40:01.845059Z","shell.execute_reply":"2021-06-03T17:40:02.058006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero_crossings = librosa.zero_crossings(data[n0:n1], pad=False)\nprint(sum(zero_crossings)) ","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:02.779800Z","iopub.execute_input":"2021-06-03T17:40:02.780257Z","iopub.status.idle":"2021-06-03T17:40:02.791227Z","shell.execute_reply.started":"2021-06-03T17:40:02.780218Z","shell.execute_reply":"2021-06-03T17:40:02.789636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mel-Frequency Cepstral Coefficients (MFCCs)\n\nmfccs = librosa.feature.mfcc(data, sr=sample_rate)\n\n#Displaying  the MFCCs:\nplt.figure(figsize=(15, 7))\nlibrosa.display.specshow(mfccs, sr=sample_rate, x_axis='time')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:03.609683Z","iopub.execute_input":"2021-06-03T17:40:03.610098Z","iopub.status.idle":"2021-06-03T17:40:03.920719Z","shell.execute_reply.started":"2021-06-03T17:40:03.610065Z","shell.execute_reply":"2021-06-03T17:40:03.919388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Chrome features\n\nhop_length=512\nchromagram = librosa.feature.chroma_stft(data, sr=sample_rate, hop_length=hop_length)\nplt.figure(figsize=(20, 8))\nlibrosa.display.specshow(chromagram, x_axis='time', y_axis='chroma', hop_length=hop_length, cmap='coolwarm')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:04.834694Z","iopub.execute_input":"2021-06-03T17:40:04.835132Z","iopub.status.idle":"2021-06-03T17:40:05.359233Z","shell.execute_reply.started":"2021-06-03T17:40:04.835097Z","shell.execute_reply":"2021-06-03T17:40:05.357866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are total {} species\".format(train_meta['primary_label'].nunique()))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:05.774596Z","iopub.execute_input":"2021-06-03T17:40:05.775274Z","iopub.status.idle":"2021-06-03T17:40:05.793359Z","shell.execute_reply.started":"2021-06-03T17:40:05.775167Z","shell.execute_reply":"2021-06-03T17:40:05.791793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting author with maximum entries\ndef plotbar(series, pal):\n    plt.figure(figsize=(20, 9))\n    chart = sns.barplot(x=series.index, y=series.values, edgecolor=(0,0,0), linewidth=2, palette=(pal))\n    chart.set_xticklabels(chart.get_xticklabels(), rotation=45)\nspecies = train_meta['primary_label'].value_counts()[:100]\nplotbar(species, \"Blues_r\")\nauthors = train_meta['author'].value_counts()[:10]\nplotbar(authors, \"YlOrBr_r\") # series, palette","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:07.979747Z","iopub.execute_input":"2021-06-03T17:40:07.980186Z","iopub.status.idle":"2021-06-03T17:40:09.800406Z","shell.execute_reply.started":"2021-06-03T17:40:07.980151Z","shell.execute_reply":"2021-06-03T17:40:09.799154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA","metadata":{}},{"cell_type":"code","source":"filename = '../input/birdclef-2021/train_short_audio/acafly/XC11209.ogg'\nplt.figure(figsize=(18, 5))\n\n# by default librosa.load returns a sample rate of 22050\n# librosa converts input to mono, hence always \nsig, sample_rate = librosa.load(filename)\nlibrosa.display.waveplot(data, sr=sample_rate)\nprint(\"Sample Rate: \", sample_rate)\nipd.Audio(filename)\n# First, compute the spectrogram using the \"short-time Fourier transform\" (stft)\nspec = librosa.stft(sig)\n\n# Scale the amplitudes according to the decibel scale\nspec_db = librosa.amplitude_to_db(spec, ref=np.max)\n\n# Plot the spectrogram\nplt.figure(figsize=(15, 5))\nlibrosa.display.specshow(spec_db, \n                         sr=32000, \n                         x_axis='time', \n                         y_axis='hz', \n                         cmap=plt.get_cmap('viridis'))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:10.604450Z","iopub.execute_input":"2021-06-03T17:40:10.604803Z","iopub.status.idle":"2021-06-03T17:40:12.064937Z","shell.execute_reply.started":"2021-06-03T17:40:10.604773Z","shell.execute_reply":"2021-06-03T17:40:12.063601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> In the past that convolutional neural networks (CNN) perform particularly well for sound classification. But CNN need 2D inputs. Luckily, we can transform an audio signal into a 2D representation: a so-called spectrogram.**","metadata":{}},{"cell_type":"code","source":"train_labels['audio_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:40:14.029680Z","iopub.execute_input":"2021-06-03T17:40:14.030048Z","iopub.status.idle":"2021-06-03T17:40:14.038792Z","shell.execute_reply.started":"2021-06-03T17:40:14.030018Z","shell.execute_reply":"2021-06-03T17:40:14.037417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.groupby(by=['audio_id']).count()['birds'][:4]","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:38.940371Z","iopub.execute_input":"2021-06-03T17:41:38.940787Z","iopub.status.idle":"2021-06-03T17:41:38.955282Z","shell.execute_reply.started":"2021-06-03T17:41:38.940754Z","shell.execute_reply":"2021-06-03T17:41:38.954170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('original label:', train_labels.loc[458, 'birds'])\nprint('split into list:', train_labels.loc[458, 'birds'].split(' '))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:39.350208Z","iopub.execute_input":"2021-06-03T17:41:39.350886Z","iopub.status.idle":"2021-06-03T17:41:39.359926Z","shell.execute_reply.started":"2021-06-03T17:41:39.350844Z","shell.execute_reply":"2021-06-03T17:41:39.358087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\nfor row in train_labels.index:\n    labels.extend(train_labels.loc[row, 'birds'].split(' '))\nlabels = list(set(labels))\n\nprint('Number of unique bird labels:', len(labels))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:40.094888Z","iopub.execute_input":"2021-06-03T17:41:40.095324Z","iopub.status.idle":"2021-06-03T17:41:40.542245Z","shell.execute_reply.started":"2021-06-03T17:41:40.095281Z","shell.execute_reply":"2021-06-03T17:41:40.541308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels_train = pd.DataFrame(index=train_labels.index, columns=labels)\nfor row in train_labels.index:\n    birds = train_labels.loc[row, 'birds'].split(' ')\n    for bird in birds:\n        df_labels_train.loc[row, bird] = 1\ndf_labels_train.fillna(0, inplace=True)\n\n# We set a dummy value for the target label in the test data because we will need for the Data Generator\ntest_data['birds'] = 'nocall'\n\ndf_labels_test = pd.DataFrame(index=test_data.index, columns=labels)\nfor row in test_data.index:\n    birds = test_data.loc[row, 'birds'].split(' ')\n    for bird in birds:\n        df_labels_test.loc[row, bird] = 1\ndf_labels_test.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:40.604955Z","iopub.execute_input":"2021-06-03T17:41:40.605637Z","iopub.status.idle":"2021-06-03T17:41:40.950440Z","shell.execute_reply.started":"2021-06-03T17:41:40.605597Z","shell.execute_reply":"2021-06-03T17:41:40.949581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels_train.sum().sort_values(ascending=False)[:10]","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:41.134337Z","iopub.execute_input":"2021-06-03T17:41:41.134832Z","iopub.status.idle":"2021-06-03T17:41:41.147529Z","shell.execute_reply.started":"2021-06-03T17:41:41.134798Z","shell.execute_reply":"2021-06-03T17:41:41.146874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.concat([train_labels, df_labels_train], axis=1)\ntest_data = pd.concat([test_data, df_labels_test], axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:41.724356Z","iopub.execute_input":"2021-06-03T17:41:41.724712Z","iopub.status.idle":"2021-06-03T17:41:41.736016Z","shell.execute_reply.started":"2021-06-03T17:41:41.724681Z","shell.execute_reply":"2021-06-03T17:41:41.735028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file = os.listdir(path+'train_soundscapes')[0]\nfile","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:42.319699Z","iopub.execute_input":"2021-06-03T17:41:42.320140Z","iopub.status.idle":"2021-06-03T17:41:42.331235Z","shell.execute_reply.started":"2021-06-03T17:41:42.320094Z","shell.execute_reply":"2021-06-03T17:41:42.329843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, samplerate = read_ogg_file(path+'train_soundscapes/', file)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:42.874937Z","iopub.execute_input":"2021-06-03T17:41:42.875351Z","iopub.status.idle":"2021-06-03T17:41:43.744563Z","shell.execute_reply.started":"2021-06-03T17:41:42.875317Z","shell.execute_reply":"2021-06-03T17:41:43.743425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_id = file.split('_')[0]\nsite = file.split('_')[1]\nprint('audio_id:', audio_id, ', site:', site)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:43.746120Z","iopub.execute_input":"2021-06-03T17:41:43.746436Z","iopub.status.idle":"2021-06-03T17:41:43.752918Z","shell.execute_reply.started":"2021-06-03T17:41:43.746407Z","shell.execute_reply":"2021-06-03T17:41:43.751869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels[(train_labels['audio_id']==int(audio_id)) & (train_labels['site']==site) & (train_labels['birds']!='nocall')]","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:41:44.064690Z","iopub.execute_input":"2021-06-03T17:41:44.065072Z","iopub.status.idle":"2021-06-03T17:41:44.111387Z","shell.execute_reply.started":"2021-06-03T17:41:44.065033Z","shell.execute_reply":"2021-06-03T17:41:44.110115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_data = data[int(455/5)*160000:int(460/5)*160000]\n","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:42:35.740820Z","iopub.execute_input":"2021-06-03T17:42:35.741285Z","iopub.status.idle":"2021-06-03T17:42:35.747018Z","shell.execute_reply.started":"2021-06-03T17:42:35.741247Z","shell.execute_reply":"2021-06-03T17:42:35.745863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 5))\nlibrosa.display.waveplot(sub_data, sr=samplerate)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:42:36.614464Z","iopub.execute_input":"2021-06-03T17:42:36.614876Z","iopub.status.idle":"2021-06-03T17:42:36.841295Z","shell.execute_reply.started":"2021-06-03T17:42:36.614838Z","shell.execute_reply":"2021-06-03T17:42:36.840150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display.Audio(sub_data, rate=samplerate)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:21.035410Z","iopub.execute_input":"2021-06-03T17:43:21.035821Z","iopub.status.idle":"2021-06-03T17:43:21.068439Z","shell.execute_reply.started":"2021-06-03T17:43:21.035788Z","shell.execute_reply":"2021-06-03T17:43:21.067102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_lenght = 160000\naudio_lenght = 5\nnum_labels = len(labels)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:21.864570Z","iopub.execute_input":"2021-06-03T17:43:21.864960Z","iopub.status.idle":"2021-06-03T17:43:21.869268Z","shell.execute_reply.started":"2021-06-03T17:43:21.864926Z","shell.execute_reply":"2021-06-03T17:43:21.868358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 16","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:22.489412Z","iopub.execute_input":"2021-06-03T17:43:22.489782Z","iopub.status.idle":"2021-06-03T17:43:22.494627Z","shell.execute_reply.started":"2021-06-03T17:43:22.489749Z","shell.execute_reply":"2021-06-03T17:43:22.493176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_IDs_train, list_IDs_val = train_test_split(list(train_labels.index), test_size=0.33, random_state=2021)\nlist_IDs_test = list(samp_subm.index)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:23.284755Z","iopub.execute_input":"2021-06-03T17:43:23.285132Z","iopub.status.idle":"2021-06-03T17:43:23.292907Z","shell.execute_reply.started":"2021-06-03T17:43:23.285101Z","shell.execute_reply":"2021-06-03T17:43:23.291481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, data, batch_size):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.data = data\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(self.list_IDs))\n        \n    def __len__(self):\n        len_ = int(len(self.list_IDs)/self.batch_size)\n        if len_*self.batch_size < len(self.list_IDs):\n            len_ += 1\n        return len_\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        X = X.reshape((self.batch_size, 100, 1600//2))\n        return X, y\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.zeros((self.batch_size, data_lenght//2))\n        y = np.zeros((self.batch_size, num_labels))\n        for i, ID in enumerate(list_IDs_temp):\n            prefix = str(self.data.loc[ID, 'audio_id'])+'_'+self.data.loc[ID, 'site']\n            file_list = [s for s in os.listdir(self.path) if prefix in s]\n            if len(file_list) == 0:\n                # Dummy for missing test audio files\n                audio_file_fft = np.zeros((data_lenght//2))\n            else:\n                file = file_list[0]#[s for s in os.listdir(self.path) if prefix in s][0]\n                audio_file, audio_sr = read_ogg_file(self.path, file)\n                audio_file = audio_file[int((self.data.loc[ID, 'seconds']-5)/audio_lenght)*data_lenght:int(self.data.loc[ID, 'seconds']/audio_lenght)*data_lenght]\n                audio_file_fft = np.abs(np.fft.fft(audio_file)[: len(audio_file)//2])\n                # scale data\n                audio_file_fft = (audio_file_fft-audio_file_fft.mean())/audio_file_fft.std()\n        X[i, ] = audio_file_fft\n        y[i, ] = self.data.loc[ID, self.data.columns[5:]].values\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:24.314965Z","iopub.execute_input":"2021-06-03T17:43:24.315434Z","iopub.status.idle":"2021-06-03T17:43:24.333167Z","shell.execute_reply.started":"2021-06-03T17:43:24.315396Z","shell.execute_reply":"2021-06-03T17:43:24.331582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = DataGenerator(path+'train_soundscapes/', list_IDs_train, train_labels, batch_size)\nval_generator = DataGenerator(path+'train_soundscapes/', list_IDs_val, train_labels, batch_size)\ntest_generator = DataGenerator(path+'test_soundscapes/', list_IDs_test, test_data, batch_size)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:25.130131Z","iopub.execute_input":"2021-06-03T17:43:25.130563Z","iopub.status.idle":"2021-06-03T17:43:25.136686Z","shell.execute_reply.started":"2021-06-03T17:43:25.130527Z","shell.execute_reply":"2021-06-03T17:43:25.135282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 2\nlernrate = 1e-3","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:25.879650Z","iopub.execute_input":"2021-06-03T17:43:25.880020Z","iopub.status.idle":"2021-06-03T17:43:25.885050Z","shell.execute_reply.started":"2021-06-03T17:43:25.879988Z","shell.execute_reply":"2021-06-03T17:43:25.883675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels\n","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:27.309576Z","iopub.execute_input":"2021-06-03T17:43:27.309945Z","iopub.status.idle":"2021-06-03T17:43:27.315672Z","shell.execute_reply.started":"2021-06-03T17:43:27.309916Z","shell.execute_reply":"2021-06-03T17:43:27.314930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model 1 - FFT based 1D CNN Neural Network****","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv1D(128, input_shape=(100, 1600//2,), kernel_size=5, strides=4, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool1D(pool_size=(4)))\nmodel.add(Conv1D(64, kernel_size=3, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dense(num_labels, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:43:43.995598Z","iopub.execute_input":"2021-06-03T17:43:43.996191Z","iopub.status.idle":"2021-06-03T17:43:44.194469Z","shell.execute_reply.started":"2021-06-03T17:43:43.996154Z","shell.execute_reply":"2021-06-03T17:43:44.193155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer = Adam(lr=lernrate),\n              loss='binary_crossentropy',\n              metrics=['binary_accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:44:11.095539Z","iopub.execute_input":"2021-06-03T17:44:11.095932Z","iopub.status.idle":"2021-06-03T17:44:11.116232Z","shell.execute_reply.started":"2021-06-03T17:44:11.095897Z","shell.execute_reply":"2021-06-03T17:44:11.114826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:44:16.409461Z","iopub.execute_input":"2021-06-03T17:44:16.409867Z","iopub.status.idle":"2021-06-03T17:44:16.420721Z","shell.execute_reply.started":"2021-06-03T17:44:16.409826Z","shell.execute_reply":"2021-06-03T17:44:16.419213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(generator=train_generator, validation_data=val_generator, epochs = epochs, workers=4)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T17:44:21.514309Z","iopub.execute_input":"2021-06-03T17:44:21.514671Z","iopub.status.idle":"2021-06-03T18:12:30.175376Z","shell.execute_reply.started":"2021-06-03T17:44:21.514641Z","shell.execute_reply":"2021-06-03T18:12:30.174108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 2, figsize=(16, 4))\nfig.subplots_adjust(hspace = .2, wspace=.2)\naxs = axs.ravel()\nloss = history.history['loss']\nloss_val = history.history['val_loss']\nepochs = range(1, len(loss)+1)\naxs[0].plot(epochs, loss, 'bo', label='loss_train')\naxs[0].plot(epochs, loss_val, 'ro', label='loss_val')\naxs[0].set_title('Value of the loss function')\naxs[0].set_xlabel('epochs')\naxs[0].set_ylabel('value of the loss function')\naxs[0].legend()\naxs[0].grid()\nacc = history.history['binary_accuracy']\nacc_val = history.history['val_binary_accuracy']\naxs[1].plot(epochs, acc, 'bo', label='accuracy_train')\naxs[1].plot(epochs, acc_val, 'ro', label='accuracy_val')\naxs[1].set_title('Accuracy')\naxs[1].set_xlabel('Epochs')\naxs[1].set_ylabel('Value of accuracy')\naxs[1].legend()\naxs[1].grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:17:47.787471Z","iopub.execute_input":"2021-06-03T18:17:47.787965Z","iopub.status.idle":"2021-06-03T18:17:48.187289Z","shell.execute_reply.started":"2021-06-03T18:17:47.787916Z","shell.execute_reply":"2021-06-03T18:17:48.186438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict_generator(test_generator, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:17:53.045126Z","iopub.execute_input":"2021-06-03T18:17:53.045921Z","iopub.status.idle":"2021-06-03T18:17:53.290780Z","shell.execute_reply.started":"2021-06-03T18:17:53.045877Z","shell.execute_reply":"2021-06-03T18:17:53.289604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = np.where(y_pred > 0.5, 1, 0)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:17:56.180060Z","iopub.execute_input":"2021-06-03T18:17:56.180499Z","iopub.status.idle":"2021-06-03T18:17:56.186614Z","shell.execute_reply.started":"2021-06-03T18:17:56.180464Z","shell.execute_reply":"2021-06-03T18:17:56.185373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for row in samp_subm.index:\n    string = ''\n    for col in range(len(y_test[row])):\n        if y_test[row][col] == 1:\n            if string == '':\n                string += labels[col]\n            else:\n                string += ' ' + labels[col]\n    if string == '':\n        string = 'nocall'\n    samp_subm.loc[row, 'birds'] = string","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:18:00.660168Z","iopub.execute_input":"2021-06-03T18:18:00.660648Z","iopub.status.idle":"2021-06-03T18:18:00.671652Z","shell.execute_reply.started":"2021-06-03T18:18:00.660605Z","shell.execute_reply":"2021-06-03T18:18:00.669748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = samp_subm\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:18:02.845078Z","iopub.execute_input":"2021-06-03T18:18:02.845501Z","iopub.status.idle":"2021-06-03T18:18:02.855840Z","shell.execute_reply.started":"2021-06-03T18:18:02.845459Z","shell.execute_reply":"2021-06-03T18:18:02.854803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Model 2 - Spectrums based 2D CNN****","metadata":{}},{"cell_type":"code","source":"import os\nimport warnings\nimport shutil\nwarnings.filterwarnings(action='ignore')\n\nimport math\nimport pandas as pd\nimport librosa\nimport librosa.display\nimport numpy as np\nimport seaborn as sns; sns.set(style='whitegrid')\nimport matplotlib.pyplot as plt\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm,tnrange,tqdm_notebook\nimport tensorflow as tf\nfrom tqdm.keras import TqdmCallback\nfrom keras.callbacks import ReduceLROnPlateau,EarlyStopping,ModelCheckpoint\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator \nfrom tensorflow.keras import applications as app\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten,AveragePooling2D\nfrom tensorflow.keras.layers import Dense,BatchNormalization,Dropout\nfrom tensorflow.keras.models import Sequential \nfrom tensorflow.keras.applications import EfficientNetB4, ResNet50,ResNet101, VGG16, MobileNet, InceptionV3","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:18:10.666135Z","iopub.execute_input":"2021-06-03T18:18:10.666744Z","iopub.status.idle":"2021-06-03T18:18:10.686135Z","shell.execute_reply.started":"2021-06-03T18:18:10.666694Z","shell.execute_reply":"2021-06-03T18:18:10.685011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Global Coefficients that can be modified\nclass coefs:\n    \n    # Generate Subset\n    rat_id = 4 # rating subset limiter \n    recs = 200 # each specie must have X recodings\n    max_files = 1500 # general last limit for rows\n    thresh = 0.25 # label probability selection threshold\n    submission = True # For Submission Only (Less Inference Output)\n    \n    # Global vars\n    seed = 1337\n    sr = 32000        # librosa sample rate input\n    sl = 5 # seconds   \n    sshape = (48,128) # height x width\n    fmin = 500      # spectrum min frequency\n    fmax = 12500    # spectrum max frequency\n    n_epoch = 100   # training epochs\n    cutoff = 15     # 3 sample spectogram (training) overwritten for inference\npath_switch = False","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:18:15.476292Z","iopub.execute_input":"2021-06-03T18:18:15.476799Z","iopub.status.idle":"2021-06-03T18:18:15.483396Z","shell.execute_reply.started":"2021-06-03T18:18:15.476765Z","shell.execute_reply":"2021-06-03T18:18:15.482598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #Plot Keras Training History\ndef HistPlot():\n\n    fig,ax = plt.subplots(1,2,figsize=(12,4))\n    sns.despine(top=True,left=True,bottom=True)\n\n    ax[0].plot(history.history['accuracy'])\n    ax[0].plot(history.history['val_accuracy'])\n    ax[0].set_title('model accuracy')\n    ax[0].set_ylabel('accuracy')\n    ax[0].set_xlabel('epoch')\n    ax[0].grid(True,linestyle='--',alpha=0.5)\n    \n    ax[1].plot(history.history['loss'])\n    ax[1].plot(history.history['val_loss'])\n    ax[1].set_title('model loss')\n    ax[1].set_ylabel('loss')\n    ax[1].set_xlabel('epoch')\n    ax[1].legend(['train', 'test'], loc='upper left')\n    ax[1].grid(True,linestyle='--',alpha=0.5)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:18:16.671431Z","iopub.execute_input":"2021-06-03T18:18:16.672005Z","iopub.status.idle":"2021-06-03T18:18:16.683738Z","shell.execute_reply.started":"2021-06-03T18:18:16.671947Z","shell.execute_reply":"2021-06-03T18:18:16.682423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the Input signal into segments\ndef split_signal(sig):\n    sig_splits = []\n    for i in range(0, len(sig), int(coefs.sl * coefs.sr)):\n        split = sig[i:i + int(coefs.sl * coefs.sr)]\n        if len(split) < int(coefs.sl * coefs.sr):\n            break\n        sig_splits.append(split)\n    \n    return sig_splits\n\n# extracts spectrograms and saves them in a working directory\ndef get_spectrograms(filepath, primary_label, output_dir):\n\n    # duration is set from global variable\n    sig, rate = librosa.load(filepath, sr=coefs.sr, offset=None, duration=coefs.cutoff)\n    sig_splits = split_signal(sig) # split the signal into parts\n    \n    # Extract mel spectrograms for each audio chunk\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n        \n        hop_length = int(coefs.sl * coefs.sr / (coefs.sshape[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                  sr=coefs.sr, \n                                                  n_fft=1024, \n                                                  hop_length=hop_length, \n                                                  n_mels=coefs.sshape[0], \n                                                  fmin=coefs.fmin, \n                                                  fmax=coefs.fmax)\n    \n        mel_spec = librosa.power_to_db(mel_spec**2, ref=np.max) \n        \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] +  \n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n        \n        saved_samples.append(save_path)\n        s_cnt += 1\n        \n    return saved_samples\n\n# https://stackoverflow.com/questions/1524126/how-to-print-a-list-more-nicely\ndef list_columns(obj, cols=4, columnwise=True, gap=4):\n    sobj = [str(item) for item in obj]\n    if cols > len(sobj): cols = len(sobj)\n    max_len = max([len(item) for item in sobj])\n    if columnwise: cols = int(math.ceil(float(len(sobj)) / float(cols)))\n    plist = [sobj[i: i+cols] for i in range(0, len(sobj), cols)]\n    if columnwise:\n        if not len(plist[-1]) == cols:\n            plist[-1].extend(['']*(len(sobj) - len(plist[-1])))\n        plist = zip(*plist)\n    printer = '\\n'.join([\n        ''.join([c.ljust(max_len + gap) for c in p])\n        for p in plist])\n    print (printer)","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:18:18.167648Z","iopub.execute_input":"2021-06-03T18:18:18.168314Z","iopub.status.idle":"2021-06-03T18:18:18.186151Z","shell.execute_reply.started":"2021-06-03T18:18:18.168274Z","shell.execute_reply":"2021-06-03T18:18:18.185240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint('STEP 1) CREATING A SUBSET OF DATASET:\\n')\n\nif(path_switch):\n    lpath = '.\\\\train_metadata.csv'\nelse:\n    lpath = '../input/birdclef-2021/train_metadata.csv'\ntrain = pd.read_csv(lpath)\nprint(f\"[DATASET]: {train.values.shape} : LABELS {len(train.primary_label.value_counts())}\")\n\n# subset filter 1 (rating)\ntemp_str = 'rating>='+str(coefs.rat_id)\ntrain = train.query(temp_str)\nprint('\\nRATING LIMITER APPLIED:')\nprint(f'[SUBSET]: {train.values.shape} : LABELS {len(train.primary_label.value_counts())}')\n\n# subset filter 2 (number of recordings per specie)\nbirds_count = {};\na = train.primary_label.unique() \na_val = train.groupby('primary_label')['primary_label'].count().values\nfor bird_species, count in zip(a,a_val):\n    birds_count[bird_species] = count\nto_model_spec = [key for key,value in birds_count.items() if value >= coefs.recs] \n\nprint(f'\\n {coefs.recs}+ RECORDINGS ONLY BIRDS LIMITED:')\nTRAIN = train.query('primary_label in @to_model_spec')\nLABELS = sorted(TRAIN.primary_label.unique())\nprint(f'[SUBSET]: {TRAIN.values.shape} : LABELS {len(LABELS)}')\n\nprint('\\n BIRD LABELS AVAILABLE AFTER FILTER:')\nlist_columns(to_model_spec, cols=4, columnwise=True, gap=4)\n\n# subset filter 3 (max audio files)\n\n# Shuffle the training data and limit the number of audio files to max_files\nprint('\\nLIMITING AUDIO FILES ...')\nTRAIN = shuffle(TRAIN, random_state=coefs.seed)[:coefs.max_files]\nLABELS = sorted(TRAIN.primary_label.unique())\nprint(f'[SUBSET]: {TRAIN.values.shape} : LABELS {len(LABELS)}')","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:20:27.846278Z","iopub.execute_input":"2021-06-03T18:20:27.846677Z","iopub.status.idle":"2021-06-03T18:20:28.358283Z","shell.execute_reply.started":"2021-06-03T18:20:27.846640Z","shell.execute_reply":"2021-06-03T18:20:28.356841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using CNN approach to train spectrogram\n\n# Parse audio files and extract training samples\nif(path_switch):\n    input_dir = '.\\\\train_short_audio\\\\'\n    output_dir = '.\\\\working\\\\melspectrogram_dataset\\\\'\nelse:\n    input_dir = '../input/birdclef-2021/train_short_audio/'\n    output_dir = './/working/melspectrogram_dataset/'\n\nsamples = []\nwith tqdm_notebook(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n        \n        if row.primary_label in to_model_spec:\n            audio_file_path = os.path.join(input_dir, row.primary_label, row.filename)\n            samples += get_spectrograms(audio_file_path, row.primary_label, output_dir)\nTRAIN_SPECS = shuffle(samples, random_state=coefs.seed)\nprint('SUCCESSFULLY EXTRACTED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))\n            ","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:21:58.725797Z","iopub.execute_input":"2021-06-03T18:21:58.726313Z","iopub.status.idle":"2021-06-03T18:24:10.776653Z","shell.execute_reply.started":"2021-06-03T18:21:58.726270Z","shell.execute_reply":"2021-06-03T18:24:10.775463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating Data Generators/Loader for Keras\n\ntrain_folder = './working/melspectrogram_dataset/'\nvalid_datagen = ImageDataGenerator(rescale=1./255, validation_split=0.2)\ntrain_datagen = ImageDataGenerator(rescale=1./255, validation_split=0.2,\\\n                                   shear_range=10,fill_mode='nearest')\n\ntrain_generator = train_datagen.flow_from_directory(train_folder, \n                        target_size=(coefs.sshape[0],coefs.sshape[1]),  # target size\n                        batch_size=32, \n                        seed=42,\n                        subset = \"training\",\n                        class_mode='categorical')    # batch size\nvalidation_generator = valid_datagen.flow_from_directory(train_folder, \n                        target_size=(coefs.sshape[0],coefs.sshape[1]),  # target size\n                        batch_size=32, \n                        seed=42,\n                        subset = \"validation\",\n                        class_mode='categorical')    # batch size","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:50:35.900315Z","iopub.execute_input":"2021-06-03T18:50:35.900720Z","iopub.status.idle":"2021-06-03T18:50:36.328911Z","shell.execute_reply.started":"2021-06-03T18:50:35.900686Z","shell.execute_reply":"2021-06-03T18:50:36.328098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.random.set_seed(coefs.seed)\nmodel = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(coefs.sshape[0], coefs.sshape[1],3)),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Second conv block\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Third conv block\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(256, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),  \n    tf.keras.layers.Dense(256, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])","metadata":{"execution":{"iopub.status.busy":"2021-06-03T18:52:21.800709Z","iopub.execute_input":"2021-06-03T18:52:21.801366Z","iopub.status.idle":"2021-06-03T18:52:22.041154Z","shell.execute_reply.started":"2021-06-03T18:52:21.801327Z","shell.execute_reply":"2021-06-03T18:52:22.039928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}