{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(action='ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option('display.max_columns', None)\npd.set_option('display.max_colwidth', None)\n\nimport librosa\nimport librosa.display\n\nfrom tqdm import tqdm, tqdm_notebook\ntqdm.pandas()\n\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\n\nimport gc\nimport wave\nfrom scipy.io import wavfile\nfrom IPython.display import Audio\nfrom IPython.display import display\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_palette('Set3')\n%matplotlib inline\n\nimport plotly.express as px\nfrom plotly.offline import iplot\nimport cufflinks as cf\ncf.go_offline()\ncf.set_config_file(offline = False, world_readable = True)\n\nimport tensorflow as tf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\nimport os\nprint(os.listdir('/kaggle/input/birdclef-2021/'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train data\nPATH = '/kaggle/input/birdclef-2021/'\ntrain_data = pd.read_csv(PATH + 'train_metadata.csv')\ndisplay(train_data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train labels\ntrain_labels = pd.read_csv(PATH + 'train_soundscape_labels.csv')\ndisplay(train_labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Train Data shape={train_data.shape}')\nprint(f'Train Labels shape={train_labels.shape}')\nprint('----------------------------')\nprint(f'{train_data.info()}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Recordings Count by Year\ntrain_data['year'] = train_data['date'].apply(lambda x: x.split('-')[0])\ntrain_data['month'] = train_data['date'].apply(lambda x: x.split('-')[1])\ntrain_data['day'] = train_data['date'].apply(lambda x: x.split('-')[2])\n\ntrain_data['year'] = train_data['year'].apply(lambda x: x if x[:2] in ['19', '20'] else np.nan)\ntrain_data['year'].fillna(train_data['year'].value_counts().index[0], inplace = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train_data['year'].value_counts()\npx.bar(x = temp.index, y = temp.values, title = 'Recordings by Year', \n       labels = {'x': 'Year', 'y': 'Count'})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = pd.pivot_table(data = train_data, index = 'primary_label', columns = 'month', \n                      values = 'secondary_labels', aggfunc = 'count')\nt = temp.T.iloc[:, :351].fillna(0)\npx.line(t, title = 'Bird Recordings by Month',\n       labels = {'months': 'Months', 'value': 'Num of Recordings'},)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Audio File\nsample_audio = PATH + 'train_short_audio/rucwar/XC133150.ogg'\nsignal, sr = librosa.load(sample_audio)\n\nprint(f'Rate: {sr}')\nprint(f'Signal: {signal}')\nprint(f'Lenght: {len(signal)}')\nprint(f'Duration signal: {round(len(signal)/sr, 3)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signal, _ = librosa.load(sample_audio, sr = 44100, duration = 20)\n\n# Signal\nplt.figure(figsize=(10, 6))\nlibrosa.display.waveplot(signal)\nplt.xlabel('Time')\nplt.ylabel('Amplitude')\nplt.show()\n\n# Melspectrogram\nplt.figure(figsize=(10, 6))\nmels = librosa.feature.melspectrogram(y = signal, sr = 44100, n_mels = 256, fmax = 8000)\nlibrosa.display.specshow(librosa.power_to_db(mels, ref = np.max), x_axis = 'time', y_axis = 'mel')\nplt.title('Melspectrogram')\nplt.colorbar()\nplt.show()\n\n# Fourier Transform\nplt.figure(figsize = (10, 6))\nstft = librosa.stft(y = signal)\nstft_db = librosa.amplitude_to_db(stft)\nlibrosa.display.specshow(stft_db, x_axis = 'time', y_axis = 'hz')\nplt.title('Spectrogram - STFT')\nplt.colorbar()\nplt.show()\n\n#Log Frequency Axis\nplt.figure(figsize = (10, 6))\nlibrosa.display.specshow(stft_db, sr = 44100, x_axis = 'time', y_axis = 'log')\nplt.colorbar()\nplt.title('Log Frequency Axis')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Audio(sample_audio, rate = 44100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Birds in train_short_audio: {len(os.listdir(PATH + 'train_short_audio/'))}\")\nprint(f\"Audio files in train_soundscapes: {len(os.listdir(PATH + 'train_soundscapes/'))}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the birds and their associated audio files\naudio_path = PATH + 'train_short_audio/'\nbirds_audio = {}\nfor bird in os.listdir(audio_path):\n    birds_audio[bird] = len(os.listdir(audio_path + bird))\nbirds_df = pd.DataFrame(birds_audio.items())\nbirds_df.columns = ['Birds', 'Num_Audio']\nbirds_df = birds_df.sort_values(by = 'Num_Audio', ascending = False)\npx.bar(birds_df, x = 'Birds', y = 'Num_Audio')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bird Recording Location on World Map\nimport geopandas as gpd\ngdf = gpd.GeoDataFrame(train_data, geometry=gpd.points_from_xy(train_data.longitude,train_data.latitude)) \n\nworld = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\nfig,ax = plt.subplots(figsize=(24,12))\nworld.plot(ax=ax, color='black', edgecolor='black')\ngdf.plot(ax=ax, color='red', markersize=2)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First consider only quality data - rating > 3.5\ntrain = train_data[train_data['rating'] > 3.5]\n\ntop_birds = train['primary_label'].value_counts()[train['primary_label'].value_counts().values > 75].index\ntrain = train[train['primary_label'].isin(top_birds)]\nprint(f'Train shape={train.shape}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe with primary_label, filename and filepath\nPATH_DIR = PATH + 'train_short_audio/'\n\ndf = train[['primary_label', 'filename']].sample(frac = 1).reset_index(drop = True)\ndf['filepath'] = PATH_DIR + df['primary_label'].astype(str) + '/' + df['filename'].astype(str)\nprint(f\"Classes in sample df: {df['primary_label'].nunique()}\\n\")\ndisplay(df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transform the labels into Multi-label\ndf = pd.concat([df, pd.get_dummies(df['primary_label'])], axis = 1)\nprint(f'Dataframe shape={df.shape}\\n')\ndisplay(df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_birds = df['primary_label'].unique()\ntarget_birds.__len__()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Global variables\nbatch_size = 64\nsr = 32000\nlength = sr * 2\nprint(f'Rate: {sr}\\nLenght: {length}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract Spectrogram using Librosa\ndef get_pad_audio(data):\n    if len(data) >= length:\n        return data\n    else:\n        return np.pad(data, pad_widht=(length - len(data), 0),\n                     mode = 'constant', constant_values=(0, 0))\n\n    \ndef get_chop_audio(samples):\n    num = np.random.randint(0, len(samples) - length)\n    return samples[num: num + length]\n\n\ndef get_load_spec(audio):\n    signal, _ = librosa.load(audio, sr=sr, duration=20)\n    signal, _ = librosa.effects.trim(signal)\n    signal = get_pad_audio(signal)\n    if len(signal) > length:\n        signal = get_chop_audio(signal)\n    mels = librosa.feature.melspectrogram(y=signal, sr=sr, n_mels=256, fmin=20, fmax=sr/2.0)\n    mels_db = librosa.power_to_db(mels, ref=np.max)\n    return mels\n\n\nclass AudioDataGen(tf.keras.utils.Sequence):\n    def __init__(self, data, batch_size, shuffle=False):\n        self.data = data\n        self.labels = self.data[target_birds]\n        self.shuffle = shuffle\n        self.batch_size = batch_size\n        self.list_idx = self.data.index.values\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(float(len(self.data)) / float(self.batch_size)))\n    \n    def __getitem__(self, index):\n        batch_idx = self.indices[index * self.batch_size:(index + 1) * self.batch_size]\n        idx = [self.list_idx[k] for k in batch_idx]\n        Data = []\n        Target = []\n        for i, k in enumerate(idx):\n            audio = get_load_spec(self.data['filepath'][k])\n            Data.append(audio)\n            Target.append(self.labels.loc[k].values)\n        \n        Data = np.expand_dims(np.array(Data), -1)\n        Target = np.array(Target)\n        return Data, Target\n  \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.list_idx))\n        if self.shuffle:\n            np.random.shuffle(self.indices)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = AudioDataGen(data=df, batch_size=batch_size)\n\nfor data, label in train_gen:\n    print(data.shape)\n    print(label.shape)\n    break\n    \n    \ndel train_gen\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Input, Conv2D, BatchNormalization,Activation, MaxPool2D,\\\n                                      Flatten, Dropout, Dense, MaxPooling2D, GlobalAveragePooling2D\nfrom tensorflow.keras.utils import Sequence, to_categorical, plot_model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras import backend as K\n# from keras.callbacks.callbacks import ReduceLROnPlateau, EarlyStopping, ModelChecpoints\nfrom tensorflow.keras.losses import CategoricalCrossentropy","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a simple tf model\ndef create_model_first(input_shape=(256, 126, 1)):\n    inputs = Input(shape=input_shape)\n    x = Conv2D(96, (4, 10), padding='same')(inputs)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPool2D()(x)\n    \n    x = Conv2D(64, (4, 10), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPool2D()(x)\n    \n    x = Conv2D(48, (4, 10), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPool2D()(x)\n    \n    x = Conv2D(32, (4, 10), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPool2D()(x)\n    x = Flatten()(x)\n    \n    x = Dropout(0.5)(x)\n    x = Dense(80)(x)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n    \n    x = Dense(80)(x)\n    x = BatchNormalization()(x)\n    x = Activation(\"relu\")(x)\n    \n    outputs = Dense(len(target_birds), activation='softmax')(x)\n    model = Model(inputs=inputs, outputs=outputs)\n    model.compile(optimizer=Adam(lr=0.001), loss='categorical_crossentropy', metrics=['accuracy'])\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reduce = tf.keras.callbacks.ReduceLROnPlateau(monitor = 'val_loss', patience = 2, verbose = 1, factor = 0.5)\nearly = tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', verbose = 1, patience = 5)\ncheck = tf.keras.callbacks.ModelCheckpoint(filepath = 'clef_model.h5', monitor = 'val_loss', verbose = 0, \n                                           save_best_only = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model_first()\nmodel.summary()\nplot_model(model, to_file='model.png')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data\ntrain_df, valid_df = train_test_split(df, test_size = 0.2, random_state = 2021)\nprint(f'train df shape={train_df.shape}\\nvalid df shape={valid_df.shape}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps_epoch = len(train_df) // batch_size\n\ntraingen = AudioDataGen(data = train_df, batch_size = batch_size)\nvalidgen = AudioDataGen(data = valid_df, batch_size = batch_size)\n\ndel train_df, valid_df\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train model\nhistory = model.fit(\n                traingen, \n                epochs = 5,\n                verbose = 1,\n                callbacks = [check, reduce, early],\n                steps_per_epoch = steps_epoch,\n                validation_data = validgen\n        )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('./best_model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}