{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# FORK of [ https://www.kaggle.com/stefankahl/birdclef2021-model-training ]. ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:18:37.612674Z","iopub.execute_input":"2021-12-01T07:18:37.613258Z","iopub.status.idle":"2021-12-01T07:18:37.618867Z","shell.execute_reply.started":"2021-12-01T07:18:37.613143Z","shell.execute_reply":"2021-12-01T07:18:37.617464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inference notebook** [ https://www.kaggle.com/virajkadam/birdclef-inference ]","metadata":{}},{"cell_type":"markdown","source":"# Model training\n\nIn this notebook, we will train our first model and apply this model to a soundscape. We will keep the amount of training samples, species and soundscapes to a minimum to keep the execution time as short as possible. Remember, this is only a sample implementation, feel free to explore your own workflow.\n\nThese are the steps that we will cover:\n\n\n* select audio files we want to use for training  \n* extract spectrograms from those files and save them in a working directory  \n* load selected samples into a large in-memory dataset  \n* build a simple beginners CNN  \n* train the model  \n* apply the model to a selected soundscape and look at the results \n\n# 1. Settings and imports\n\nLet’s begin with imports and a few basic settings.","metadata":{}},{"cell_type":"code","source":"import os\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nimport pandas as pd\nimport librosa\nimport numpy as np\nimport pickle\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow_addons.metrics import F1Score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-01T07:18:37.620473Z","iopub.execute_input":"2021-12-01T07:18:37.620877Z","iopub.status.idle":"2021-12-01T07:18:46.138871Z","shell.execute_reply.started":"2021-12-01T07:18:37.620849Z","shell.execute_reply":"2021-12-01T07:18:46.137716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-12-01T11:56:52.277018Z","iopub.execute_input":"2021-12-01T11:56:52.277505Z","iopub.status.idle":"2021-12-01T11:56:53.169839Z","shell.execute_reply.started":"2021-12-01T11:56:52.277422Z","shell.execute_reply":"2021-12-01T11:56:53.168828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Global vars\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5 # seconds\nSPEC_SHAPE = (48, 128) # height x width\nFMIN = 500\nFMAX = 12500\n# MAX_AUDIO_FILES = 10000\nEPOCHS=10","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:18:46.140388Z","iopub.execute_input":"2021-12-01T07:18:46.1408Z","iopub.status.idle":"2021-12-01T07:18:46.146716Z","shell.execute_reply.started":"2021-12-01T07:18:46.140748Z","shell.execute_reply":"2021-12-01T07:18:46.145539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Data preparation\n\nTaking audio files rated >=4 and having more than 175 samples.","metadata":{}},{"cell_type":"code","source":"# Code adapted from: \n# https://www.kaggle.com/frlemarchand/bird-song-classification-using-an-efficientnet\n# Make sure to check out the entire notebook.\n\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2021/train_metadata.csv',)\n\n# Limit the number of training samples and classes\n# First, only use high quality samples\ntrain = train.query('rating>=4')\n\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), \n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 175] \n\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:18:46.148763Z","iopub.execute_input":"2021-12-01T07:18:46.149253Z","iopub.status.idle":"2021-12-01T07:18:46.690927Z","shell.execute_reply.started":"2021-12-01T07:18:46.149209Z","shell.execute_reply":"2021-12-01T07:18:46.690071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving labels \nwith open('LABELS.pkl','wb') as f:\n    pickle.dump(LABELS,f)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:18:46.694136Z","iopub.execute_input":"2021-12-01T07:18:46.694474Z","iopub.status.idle":"2021-12-01T07:18:46.699052Z","shell.execute_reply.started":"2021-12-01T07:18:46.694443Z","shell.execute_reply":"2021-12-01T07:18:46.697864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# extracting spectrograms","metadata":{}},{"cell_type":"code","source":"# Shuffle the training data and limit the number of audio files to MAX_AUDIO_FILES\nTRAIN = shuffle(TRAIN, random_state=RANDOM_SEED)\n\n# Define a function that splits an audio file, \n# extracts spectrograms and saves them in a working directory\ndef get_spectrograms(filepath, primary_label, output_dir):\n    \n    # Open the file with librosa (limited to the first 15 seconds)\n    sig, rate = librosa.load(filepath, sr=SAMPLE_RATE, offset=None, duration=15)\n    \n    # Split signal into five second chunks\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n        # End of signal?\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n        \n        sig_splits.append(split)\n        \n    # Extract mel spectrograms for each audio chunk\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n        \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                  sr=SAMPLE_RATE, \n                                                  n_fft=1024, \n                                                  hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], \n                                                  fmin=FMIN, \n                                                  fmax=FMAX)\n    \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] + \n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n        \n        saved_samples.append(save_path)\n        s_cnt += 1\n        \n        \n    return saved_samples\n\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:18:46.700394Z","iopub.execute_input":"2021-12-01T07:18:46.700716Z","iopub.status.idle":"2021-12-01T07:18:46.72055Z","shell.execute_reply.started":"2021-12-01T07:18:46.700686Z","shell.execute_reply":"2021-12-01T07:18:46.719501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse audio files and extract training samples\ninput_dir = '../input/birdclef-2021/train_short_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\nsamples = []\nwith tqdm(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n        \n        if row.primary_label in most_represented_birds:\n            audio_file_path = os.path.join(input_dir, row.primary_label, row.filename)\n            samples += get_spectrograms(audio_file_path, row.primary_label, output_dir)\n            \nTRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\nprint('SUCCESSFULLY EXTRACTED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:18:46.722003Z","iopub.execute_input":"2021-12-01T07:18:46.722379Z","iopub.status.idle":"2021-12-01T07:31:54.511197Z","shell.execute_reply.started":"2021-12-01T07:18:46.722345Z","shell.execute_reply":"2021-12-01T07:31:54.509914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the first 12 spectrograms of TRAIN_SPECS\nplt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(TRAIN_SPECS[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(TRAIN_SPECS[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:31:54.513804Z","iopub.execute_input":"2021-12-01T07:31:54.514158Z","iopub.status.idle":"2021-12-01T07:31:55.880563Z","shell.execute_reply.started":"2021-12-01T07:31:54.514121Z","shell.execute_reply":"2021-12-01T07:31:55.879577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Nice! These are good samples. Notice how some of them only contain a fraction of a bird call? That's an issue we won't deal with in this tutorial. We will simply ignore the fact that samples might not contain any bird sounds.\n\n# 4. Load training samples\n\nFor now, our spectrograms reside in a working directory. If we want to train a model, we have to load them into memory. Yet, with potentially hundreds of thousands of extracted spectrograms, an in-memory dataset is not a good idea. But for now, loading samples from disk and combining them into a large NumPy array is fine. It’s the easiest way to use these data for training with Keras.","metadata":{}},{"cell_type":"code","source":"# Parse all samples and add spectrograms into train data, primary_labels into label data\ntrain_specs, train_labels = [], []\nwith tqdm(total=len(TRAIN_SPECS)) as pbar:\n    for path in TRAIN_SPECS:\n        pbar.update(1)\n\n        # Open image\n        spec = Image.open(path)\n\n        # Convert to numpy array\n        spec = np.array(spec, dtype='float32')\n        \n        # Normalize between 0.0 and 1.0\n        # and exclude samples with nan \n        spec -= spec.min()\n        spec /= spec.max()\n        if not spec.max() == 1.0 or not spec.min() == 0.0:\n            continue\n\n        # Add channel axis to 2D array\n        spec = np.expand_dims(spec, -1)\n\n        # Add new dimension for batch size\n        spec = np.expand_dims(spec, 0)\n\n        # Add to train data\n        if len(train_specs) == 0:\n            train_specs = spec\n        else:\n            train_specs = np.vstack((train_specs, spec))\n\n        # Add to label data\n        target = np.zeros((len(LABELS)), dtype='float32')\n        bird = path.split(os.sep)[-2]\n        target[LABELS.index(bird)] = 1.0\n        if len(train_labels) == 0:\n            train_labels = target\n        else:\n            train_labels = np.vstack((train_labels, target))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T07:31:55.882054Z","iopub.execute_input":"2021-12-01T07:31:55.882609Z","iopub.status.idle":"2021-12-01T08:56:15.313472Z","shell.execute_reply.started":"2021-12-01T07:31:55.88257Z","shell.execute_reply":"2021-12-01T08:56:15.31215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Build a simple model\n\n","metadata":{}},{"cell_type":"code","source":"f1_score=F1Score(num_classes=len(LABELS),average='macro',name='f1_score')","metadata":{"execution":{"iopub.status.busy":"2021-12-01T08:56:15.316316Z","iopub.execute_input":"2021-12-01T08:56:15.316809Z","iopub.status.idle":"2021-12-01T08:56:15.449907Z","shell.execute_reply.started":"2021-12-01T08:56:15.316759Z","shell.execute_reply":"2021-12-01T08:56:15.4488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 1","metadata":{}},{"cell_type":"code","source":"# Make sure your experiments are reproducible\ntf.random.set_seed(RANDOM_SEED)\n\n# Build a simple model as a sequence of  convolutional blocks.\n# Each block has the sequence CONV --> RELU --> BNORM --> MAXPOOL.\n# Finally, perform global average pooling and add 2 dense layers.\n# The last layer is our classification layer and is softmax activated.\n# (Well it's a multi-label task so sigmoid might actually be a better choice)\nmodel = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1)),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Second conv block\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Third conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(256, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n        \n    \n    \n#     # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(256, activation='relu'),  \n    tf.keras.layers.Dropout(0.5),\n#     tf.keras.layers.Dense(64, activation='relu'),   \n#     tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])\n\n\n\nprint('MODEL HAS {} PARAMETERS.'.format(model.count_params()))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T08:56:15.451398Z","iopub.execute_input":"2021-12-01T08:56:15.451912Z","iopub.status.idle":"2021-12-01T08:56:15.681893Z","shell.execute_reply.started":"2021-12-01T08:56:15.451879Z","shell.execute_reply":"2021-12-01T08:56:15.681153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model and specify optimizer, loss and metric\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy',f1_score])","metadata":{"execution":{"iopub.status.busy":"2021-12-01T08:56:15.683293Z","iopub.execute_input":"2021-12-01T08:56:15.683828Z","iopub.status.idle":"2021-12-01T08:56:15.706895Z","shell.execute_reply.started":"2021-12-01T08:56:15.683793Z","shell.execute_reply":"2021-12-01T08:56:15.705907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add callbacks to reduce the learning rate if needed, early stopping, and checkpoint saving\ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_f1_score', \n                                                  patience=2, \n                                                  verbose=1, \n                                                  mode='max',\n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', \n                                              verbose=1,\n                                              patience=20),\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model.h5', \n                                                monitor='val_f1_score',\n                                                mode='max',\n                                                verbose=1,\n                                                save_best_only=True)]","metadata":{"execution":{"iopub.status.busy":"2021-12-01T08:56:15.708553Z","iopub.execute_input":"2021-12-01T08:56:15.708945Z","iopub.status.idle":"2021-12-01T08:56:15.717132Z","shell.execute_reply.started":"2021-12-01T08:56:15.708912Z","shell.execute_reply":"2021-12-01T08:56:15.715961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_history(history):\n    his=pd.DataFrame(history.history)\n    plt.subplots(1,2,figsize=(16,8))\n    \n    #loss:\n    plt.subplot(1,2,1)\n    plt.plot(range(len(his)),his['loss'],color='g',label='training')\n    plt.plot(range(len(his)),his['val_loss'],color='r',label='validation')\n    plt.legend()\n    plt.title('Loss')\n    \n    #accuracy\n    plt.subplot(1,2,2)\n    plt.plot(range(len(his)),his['accuracy'],color='g',label='training_acc')\n    plt.plot(range(len(his)),his['val_accuracy'],color='r',label='validation_acc')\n    \n#     #f1_score\n#     plt.plot(range(len(his)),his['f1_score'],color='steelblue',label='training_f1')\n#     plt.plot(range(len(his)),his['val_f1_score'],color='maroon',label='validation_f1')\n    \n    plt.legend()\n    plt.title('accuracy')\n    \n    plt.show()              ","metadata":{"execution":{"iopub.status.busy":"2021-12-01T08:56:15.719731Z","iopub.execute_input":"2021-12-01T08:56:15.720344Z","iopub.status.idle":"2021-12-01T08:56:15.735679Z","shell.execute_reply.started":"2021-12-01T08:56:15.720293Z","shell.execute_reply":"2021-12-01T08:56:15.734815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TRAINING","metadata":{}},{"cell_type":"code","source":"# Let's train the model for a few epochs\nhis=model.fit(train_specs, \n          train_labels,\n          batch_size=32,\n          validation_split=0.2,\n          callbacks=callbacks,\n          epochs=EPOCHS)\n\nplot_history(his)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T08:56:15.737331Z","iopub.execute_input":"2021-12-01T08:56:15.737893Z","iopub.status.idle":"2021-12-01T09:15:10.369118Z","shell.execute_reply.started":"2021-12-01T08:56:15.73786Z","shell.execute_reply":"2021-12-01T09:15:10.367597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model 2","metadata":{}},{"cell_type":"code","source":"# Make sure your experiments are reproducible\ntf.random.set_seed(RANDOM_SEED)\n\n# Build a simple model as a sequence of  convolutional blocks.\n# Each block has the sequence CONV --> RELU --> BNORM --> MAXPOOL.\n# Finally, perform global average pooling and add 2 dense layers.\n# The last layer is our classification layer and is softmax activated.\n# (Well it's a multi-label task so sigmoid might actually be a better choice)\nmodel2 = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1)),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Second conv block\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Third conv block\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n        \n    \n    \n#     # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(128, activation='relu'),  \n    tf.keras.layers.Dropout(0.5),\n#     tf.keras.layers.Dense(64, activation='relu'),   \n#     tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])\n\n\n\nprint('MODEL HAS {} PARAMETERS.'.format(model2.count_params()))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:15:10.37121Z","iopub.execute_input":"2021-12-01T09:15:10.371755Z","iopub.status.idle":"2021-12-01T09:15:10.537254Z","shell.execute_reply.started":"2021-12-01T09:15:10.371699Z","shell.execute_reply":"2021-12-01T09:15:10.536009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Compile the model and specify optimizer, loss and metric\nmodel2.compile(optimizer=tf.keras.optimizers.Adam(lr=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy',f1_score])\n\n# Add callbacks to reduce the learning rate if needed, early stopping, and checkpoint saving\ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_f1_score',\n                                                  mode='max',\n                                                  patience=2, \n                                                  verbose=1, \n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', \n                                              verbose=1,\n                                              patience=20),\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model2.h5', \n                                                mode='max',\n                                                monitor='val_f1_score',\n                                                verbose=0,\n                                                save_best_only=True)]","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:15:10.538685Z","iopub.execute_input":"2021-12-01T09:15:10.539288Z","iopub.status.idle":"2021-12-01T09:15:10.555359Z","shell.execute_reply.started":"2021-12-01T09:15:10.539242Z","shell.execute_reply":"2021-12-01T09:15:10.553786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's train the model for a few epochs\nhis2=model2.fit(train_specs, \n          train_labels,\n          batch_size=32,\n          validation_split=0.2,\n          callbacks=callbacks,\n          verbose=1,\n          epochs=EPOCHS)\n\nplot_history(his2)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:15:10.5571Z","iopub.execute_input":"2021-12-01T09:15:10.557972Z","iopub.status.idle":"2021-12-01T09:29:49.940794Z","shell.execute_reply.started":"2021-12-01T09:15:10.557919Z","shell.execute_reply":"2021-12-01T09:29:49.93987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Soundscape analysis","metadata":{}},{"cell_type":"markdown","source":"**Soundscape_analysis1**","metadata":{}},{"cell_type":"code","source":"# Load the best checkpoint\nmodel = tf.keras.models.load_model('best_model.h5')\nmodel2= tf.keras.models.load_model('best_model2.h5')\n\n# Pick a soundscape\nsoundscape_path = '../input/birdclef-2021/train_soundscapes/28933_SSW_20170408.ogg'\n\n# Open it with librosa\nsig, rate = librosa.load(soundscape_path, sr=SAMPLE_RATE)\n\n# Store results so that we can analyze them later\ndata = {'row_id': [], 'prediction': [], 'score': []}\n\n# Split signal into 5-second chunks\n# Just like we did before (well, this could actually be a seperate function)\nsig_splits = []\nfor i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n    split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n    # End of signal?\n    if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n        break\n\n    sig_splits.append(split)\n    \n# Get the spectrograms and run inference on each of them\n# This should be the exact same process as we used to\n# generate training samples!\nseconds, scnt = 0, 0\nfor chunk in sig_splits:\n    \n    # Keep track of the end time of each chunk\n    seconds += 5\n        \n    # Get the spectrogram\n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=1024, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n    # Normalize to match the value range we used during training.\n    # That's something you should always double check!\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    \n    # Add channel axis to 2D array\n    mel_spec = np.expand_dims(mel_spec, -1)\n\n    # Add new dimension for batch size\n    mel_spec = np.expand_dims(mel_spec, 0)\n    \n    # Predict\n    p = model.predict(mel_spec)[0] + model2.predict(mel_spec)[0] \n    \n    \n    # Get highest scoring species\n    idx = p.argmax()\n    species = LABELS[idx]\n    score = p[idx]\n    \n    # Prepare submission entry\n    data['row_id'].append(soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                          '_' + str(seconds))    \n    \n    # Decide if it's a \"nocall\" or a species by applying a threshold\n    if score > 0.4:\n        data['prediction'].append(species)\n        scnt += 1\n    else:\n        data['prediction'].append('nocall')\n        \n    # Add the confidence score as well\n    data['score'].append(score)\n        \nprint('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:29:49.9455Z","iopub.execute_input":"2021-12-01T09:29:49.946043Z","iopub.status.idle":"2021-12-01T09:30:02.97496Z","shell.execute_reply.started":"2021-12-01T09:29:49.94599Z","shell.execute_reply":"2021-12-01T09:30:02.973894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame\nresults = pd.DataFrame(data, columns = ['row_id', 'prediction', 'score'])\n\n# Merge with ground truth so we can inspect\ngt = pd.read_csv('../input/birdclef-2021/train_soundscape_labels.csv',)\nresults = pd.merge(gt, results, on='row_id')\n\n# Let's look at the first 50 entries\nresults.head(50)\nresults.plot()","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:30:02.97684Z","iopub.execute_input":"2021-12-01T09:30:02.977141Z","iopub.status.idle":"2021-12-01T09:30:03.24989Z","shell.execute_reply.started":"2021-12-01T09:30:02.97711Z","shell.execute_reply":"2021-12-01T09:30:03.249064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Soundscape_analysis2_2**","metadata":{}},{"cell_type":"code","source":"\n# Pick a soundscape\nsoundscape_path = '../input/birdclef-2021/train_soundscapes/7843_SSW_20170325.ogg'\n\n# Open it with librosa\nsig, rate = librosa.load(soundscape_path, sr=SAMPLE_RATE)\n\n# Store results so that we can analyze them later\ndata = {'row_id': [], 'prediction': [], 'score': []}\n\n# Split signal into 5-second chunks\n# Just like we did before (well, this could actually be a seperate function)\nsig_splits = []\nfor i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n    split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n    # End of signal?\n    if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n        break\n\n    sig_splits.append(split)\n    \n# Get the spectrograms and run inference on each of them\n# This should be the exact same process as we used to\n# generate training samples!\nseconds, scnt = 0, 0\nfor chunk in sig_splits:\n    \n    # Keep track of the end time of each chunk\n    seconds += 5\n        \n    # Get the spectrogram\n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=1024, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n    # Normalize to match the value range we used during training.\n    # That's something you should always double check!\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    \n    # Add channel axis to 2D array\n    mel_spec = np.expand_dims(mel_spec, -1)\n\n    # Add new dimension for batch size\n    mel_spec = np.expand_dims(mel_spec, 0)\n    \n    # Predict\n    p = 0.5*model.predict(mel_spec)[0] + 0.5*model2.predict(mel_spec)[0]\n    \n    \n    # Get highest scoring species\n    idx = p.argmax()\n    species = LABELS[idx]\n    score = p[idx]\n    \n    # Prepare submission entry\n    data['row_id'].append(soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                          '_' + str(seconds))    \n    \n    # Decide if it's a \"nocall\" or a species by applying a threshold\n    if score > 0.45:\n        data['prediction'].append(species)\n        scnt += 1\n    else:\n        data['prediction'].append('nocall')\n        \n    # Add the confidence score as well\n    data['score'].append(score)\n        \nprint('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:30:03.251087Z","iopub.execute_input":"2021-12-01T09:30:03.251655Z","iopub.status.idle":"2021-12-01T09:30:15.897457Z","shell.execute_reply.started":"2021-12-01T09:30:03.251619Z","shell.execute_reply":"2021-12-01T09:30:15.896432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame\nresults = pd.DataFrame(data, columns = ['row_id', 'prediction', 'score'])\n\n# Merge with ground truth so we can inspect\ngt = pd.read_csv('../input/birdclef-2021/train_soundscape_labels.csv',)\nresults = pd.merge(gt, results, on='row_id')\n\n# Let's look at the first 50 entries\nresults.head(50)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:30:15.898684Z","iopub.execute_input":"2021-12-01T09:30:15.898957Z","iopub.status.idle":"2021-12-01T09:30:15.943012Z","shell.execute_reply.started":"2021-12-01T09:30:15.898929Z","shell.execute_reply":"2021-12-01T09:30:15.942201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Soundscape_analysis_3**","metadata":{}},{"cell_type":"code","source":"# Pick a soundscape\nsoundscape_path = '../input/birdclef-2021/train_soundscapes/26709_SSW_20170701.ogg'\n\n# Open it with librosa\nsig, rate = librosa.load(soundscape_path, sr=SAMPLE_RATE)\n\n# Store results so that we can analyze them later\ndata = {'row_id': [], 'prediction': [], 'score': []}\n\n# Split signal into 5-second chunks\n# Just like we did before (well, this could actually be a seperate function)\nsig_splits = []\nfor i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n    split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n    # End of signal?\n    if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n        break\n\n    sig_splits.append(split)\n    \n# Get the spectrograms and run inference on each of them\n# This should be the exact same process as we used to\n# generate training samples!\nseconds, scnt = 0, 0\nfor chunk in sig_splits:\n    \n    # Keep track of the end time of each chunk\n    seconds += 5\n        \n    # Get the spectrogram\n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=1024, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n    # Normalize to match the value range we used during training.\n    # That's something you should always double check!\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    \n    # Add channel axis to 2D array\n    mel_spec = np.expand_dims(mel_spec, -1)\n\n    # Add new dimension for batch size\n    mel_spec = np.expand_dims(mel_spec, 0)\n    \n    # Predict\n    p = 0.5*model.predict(mel_spec)[0] + 0.5*model2.predict(mel_spec)[0]\n    \n    \n    # Get highest scoring species\n    idx = p.argmax()\n    species = LABELS[idx]\n    score = p[idx]\n    \n    # Prepare submission entry\n    data['row_id'].append(soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                          '_' + str(seconds))    \n    \n    # Decide if it's a \"nocall\" or a species by applying a threshold\n    if score > 0.3:\n        data['prediction'].append(species)\n        scnt += 1\n    else:\n        data['prediction'].append('nocall')\n        \n    # Add the confidence score as well\n    data['score'].append(score)\n        \nprint('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:30:15.944199Z","iopub.execute_input":"2021-12-01T09:30:15.944689Z","iopub.status.idle":"2021-12-01T09:30:28.581001Z","shell.execute_reply.started":"2021-12-01T09:30:15.944655Z","shell.execute_reply":"2021-12-01T09:30:28.579973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame\nresults = pd.DataFrame(data, columns = ['row_id', 'prediction', 'score'])\n\n# Merge with ground truth so we can inspect\ngt = pd.read_csv('../input/birdclef-2021/train_soundscape_labels.csv',)\nresults = pd.merge(gt, results, on='row_id')\n\n# Let's look at the first 50 entries\nresults.head(100)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T09:30:28.582336Z","iopub.execute_input":"2021-12-01T09:30:28.582644Z","iopub.status.idle":"2021-12-01T09:30:28.617264Z","shell.execute_reply.started":"2021-12-01T09:30:28.582612Z","shell.execute_reply":"2021-12-01T09:30:28.61633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:17:28.046414Z","iopub.execute_input":"2021-12-01T12:17:28.046777Z","iopub.status.idle":"2021-12-01T12:17:28.051531Z","shell.execute_reply.started":"2021-12-01T12:17:28.046746Z","shell.execute_reply":"2021-12-01T12:17:28.050452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv('../input/accuracya/bird_audio.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:18:04.857264Z","iopub.execute_input":"2021-12-01T12:18:04.857854Z","iopub.status.idle":"2021-12-01T12:18:04.873688Z","shell.execute_reply.started":"2021-12-01T12:18:04.857817Z","shell.execute_reply":"2021-12-01T12:18:04.872476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\nimport ast\n\ndata = defaultdict(list)\n\nfor idx, row in df.iterrows():\n    array = []\n    feature = row['feature']\n    for idx, num in enumerate(feature[1:-1].split()):\n        data[f'feature_{idx}'].append(float(num))\n    data['class_number'].append(row['class_number'])\ndata = pd.DataFrame(data)\ndata","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:18:07.328229Z","iopub.execute_input":"2021-12-01T12:18:07.328641Z","iopub.status.idle":"2021-12-01T12:18:07.532448Z","shell.execute_reply.started":"2021-12-01T12:18:07.328603Z","shell.execute_reply":"2021-12-01T12:18:07.531352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = data['class_number']","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:18:24.334725Z","iopub.execute_input":"2021-12-01T12:18:24.335108Z","iopub.status.idle":"2021-12-01T12:18:24.339399Z","shell.execute_reply.started":"2021-12-01T12:18:24.335074Z","shell.execute_reply":"2021-12-01T12:18:24.338431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data.drop(['class_number'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:18:39.318124Z","iopub.execute_input":"2021-12-01T12:18:39.318521Z","iopub.status.idle":"2021-12-01T12:18:39.324359Z","shell.execute_reply.started":"2021-12-01T12:18:39.318485Z","shell.execute_reply":"2021-12-01T12:18:39.323333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\nX_train, X_test , y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:18:57.285182Z","iopub.execute_input":"2021-12-01T12:18:57.285529Z","iopub.status.idle":"2021-12-01T12:18:57.294085Z","shell.execute_reply.started":"2021-12-01T12:18:57.2855Z","shell.execute_reply":"2021-12-01T12:18:57.292823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.metrics import confusion_matrix, accuracy_score\n\nsvc = SVC(kernel='linear', gamma=0.1)\nsvc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:19:10.411827Z","iopub.execute_input":"2021-12-01T12:19:10.412212Z","iopub.status.idle":"2021-12-01T12:19:12.856059Z","shell.execute_reply.started":"2021-12-01T12:19:10.412181Z","shell.execute_reply":"2021-12-01T12:19:12.85515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:19:28.298627Z","iopub.execute_input":"2021-12-01T12:19:28.299078Z","iopub.status.idle":"2021-12-01T12:19:28.303939Z","shell.execute_reply.started":"2021-12-01T12:19:28.299039Z","shell.execute_reply":"2021-12-01T12:19:28.302799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = svc.predict(X_test)\ntrain_accuracy = svc.score(X_train, y_train)\ntest_accuracy = svc.score(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:19:36.541011Z","iopub.execute_input":"2021-12-01T12:19:36.541369Z","iopub.status.idle":"2021-12-01T12:19:36.578109Z","shell.execute_reply.started":"2021-12-01T12:19:36.54134Z","shell.execute_reply":"2021-12-01T12:19:36.577136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_accuracy)\nprint(test_accuracy)","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:19:46.555099Z","iopub.execute_input":"2021-12-01T12:19:46.555452Z","iopub.status.idle":"2021-12-01T12:19:46.562756Z","shell.execute_reply.started":"2021-12-01T12:19:46.555423Z","shell.execute_reply":"2021-12-01T12:19:46.561333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction\n**Inference notebook** [ https://www.kaggle.com/virajkadam/birdclef-inference ]","metadata":{}}]}