{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pickle\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:45:56.515782Z","iopub.execute_input":"2022-01-06T08:45:56.516332Z","iopub.status.idle":"2022-01-06T08:45:56.520097Z","shell.execute_reply.started":"2022-01-06T08:45:56.516294Z","shell.execute_reply":"2022-01-06T08:45:56.519146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nimport warnings \nwarnings.filterwarnings(action=\"ignore\")\n\nimport pandas as pd\nimport librosa\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow_addons.metrics import F1Score","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:45:57.309881Z","iopub.execute_input":"2022-01-06T08:45:57.310538Z","iopub.status.idle":"2022-01-06T08:46:03.400007Z","shell.execute_reply.started":"2022-01-06T08:45:57.310503Z","shell.execute_reply":"2022-01-06T08:46:03.398929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Global Variables\nRANDOM_SEED = 42\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5\nSPEC_SHAPE = (48, 128)\nFMIN = 500\nFMAX = 12500\nEPOCHS = 50","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:46:07.608053Z","iopub.execute_input":"2022-01-06T08:46:07.60886Z","iopub.status.idle":"2022-01-06T08:46:07.617926Z","shell.execute_reply.started":"2022-01-06T08:46:07.608823Z","shell.execute_reply":"2022-01-06T08:46:07.612975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load meta-data file\ntrain = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\n#selecting files with rating greater than 4\ntrain = train.query('rating >= 4')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:46:08.309218Z","iopub.execute_input":"2022-01-06T08:46:08.309858Z","iopub.status.idle":"2022-01-06T08:46:08.777519Z","shell.execute_reply.started":"2022-01-06T08:46:08.309821Z","shell.execute_reply":"2022-01-06T08:46:08.776824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Data Preparation:\n\nbird_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), train.groupby(\"primary_label\")[\"primary_label\"].count().values):\n    bird_count[bird_species] = count\nmost_represented_birds = {bird for bird, count in bird_count.items() if count >= 175}\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\nprint(\"Number of species in train_data\", len(LABELS))\nprint(\"Number of samples in train_data\", len(TRAIN))\nprint(\"Labels:\", LABELS)\n\n#Saving Labels\nwith open('Labels.pkl', 'wb') as f:\n    pickle.dump(LABELS, f)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:46:08.904148Z","iopub.execute_input":"2022-01-06T08:46:08.904367Z","iopub.status.idle":"2022-01-06T08:46:08.939009Z","shell.execute_reply.started":"2022-01-06T08:46:08.904341Z","shell.execute_reply":"2022-01-06T08:46:08.938281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shuffle data\n\nTRAIN = shuffle(TRAIN, random_state = RANDOM_SEED)\n\n#get spectrogram image\ndef get_spectrogram(file_path, primary_label, output_dir):\n    #opening file with librosa limited to first 15 seconds\n    sig, rate = librosa.load(file_path, sr = SAMPLE_RATE, offset = None, duration = 15)\n    #split into 5 second chunks\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i+int(SIGNAL_LENGTH * SAMPLE_RATE)]\n        #end of signal\n        if(len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE)):\n            break\n        sig_splits.append(split)\n                   \n                   \n    #extract mel-spectrograms for each chunks\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n        \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                  sr=SAMPLE_RATE, \n                                                  n_fft=1024, \n                                                  hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], \n                                                  fmin=FMIN, \n                                                  fmax=FMAX)\n    \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, file_path.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] + \n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n        \n        saved_samples.append(save_path)\n        s_cnt += 1\n        \n        \n    return saved_samples\n\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))\n        \n                   ","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:46:26.210102Z","iopub.execute_input":"2022-01-06T08:46:26.210537Z","iopub.status.idle":"2022-01-06T08:46:26.232744Z","shell.execute_reply.started":"2022-01-06T08:46:26.210499Z","shell.execute_reply":"2022-01-06T08:46:26.231769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse audio files and extract training samples\ninput_dir = '../input/birdclef-2021/train_short_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\nsamples = []\nwith tqdm(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n        \n        if row.primary_label in most_represented_birds:\n            audio_file_path = os.path.join(input_dir, row.primary_label, row.filename)\n            samples += get_spectrogram(audio_file_path, row.primary_label, output_dir)\n            \nTRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\nprint('SUCCESSFULLY EXTRACTED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))","metadata":{"execution":{"iopub.status.busy":"2022-01-06T08:46:43.995182Z","iopub.execute_input":"2022-01-06T08:46:43.995453Z","iopub.status.idle":"2022-01-06T08:59:29.881149Z","shell.execute_reply.started":"2022-01-06T08:46:43.995423Z","shell.execute_reply":"2022-01-06T08:59:29.880333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot spectrograms of TRAIN_SPECS\nplt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(TRAIN_SPECS[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(TRAIN_SPECS[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2022-01-06T09:07:16.346282Z","iopub.execute_input":"2022-01-06T09:07:16.346571Z","iopub.status.idle":"2022-01-06T09:07:17.819311Z","shell.execute_reply.started":"2022-01-06T09:07:16.346542Z","shell.execute_reply":"2022-01-06T09:07:17.818602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse all samples and add spectrograms into train data, primary_labels into label data\ntrain_specs, train_labels = [], []\nwith tqdm(total=len(TRAIN_SPECS)) as pbar:\n    for path in TRAIN_SPECS:\n        pbar.update(1)\n\n        # Open image\n        spec = Image.open(path)\n\n        # Convert to numpy array\n        spec = np.array(spec, dtype='float32')\n        \n        # Normalize between 0.0 and 1.0\n        # and exclude samples with nan \n        spec -= spec.min()\n        spec /= spec.max()\n        if not spec.max() == 1.0 or not spec.min() == 0.0:\n            continue\n        \n\n        # Add channel axis to 2D array\n        spec = np.expand_dims(spec, -1)\n\n        # Add new dimension for batch size\n        spec = np.expand_dims(spec, 0)\n        \n        # Add to train data\n        if len(train_specs) == 0:\n            train_specs = spec\n        else:\n            train_specs = np.vstack((train_specs, spec))\n\n        # Add to label data\n        target = np.zeros((len(LABELS)), dtype='float32')\n        bird = path.split(os.sep)[-2]\n        target[LABELS.index(bird)] = 1.0\n        if len(train_labels) == 0:\n            train_labels = target\n        else:\n            train_labels = np.vstack((train_labels, target))","metadata":{"execution":{"iopub.status.busy":"2022-01-06T09:07:19.936328Z","iopub.execute_input":"2022-01-06T09:07:19.936581Z","iopub.status.idle":"2022-01-06T09:12:58.001051Z","shell.execute_reply.started":"2022-01-06T09:07:19.936552Z","shell.execute_reply":"2022-01-06T09:12:57.999881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score=F1Score(num_classes=len(LABELS),average='macro',name='f1_score')\n","metadata":{"execution":{"iopub.status.busy":"2021-12-28T13:29:29.49347Z","iopub.execute_input":"2021-12-28T13:29:29.494082Z","iopub.status.idle":"2021-12-28T13:29:31.707173Z","shell.execute_reply.started":"2021-12-28T13:29:29.494042Z","shell.execute_reply":"2021-12-28T13:29:31.705871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make sure your experiments are reproducible\ntf.random.set_seed(RANDOM_SEED)\n\n# Each block has the sequence CONV --> RELU --> BNORM --> MAXPOOL.\n# Finally, global average pooling and 2 dense layers.\nmodel = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1)),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Second conv block\n    tf.keras.layers.Conv2D(32, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Third conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(256, (3, 3), activation='relu'),\n    tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n        \n    \n    \n#     # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(256, activation='relu'),  \n    tf.keras.layers.Dropout(0.5),\n#     tf.keras.layers.Dense(64, activation='relu'),   \n#     tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])\n\n\n\nprint('MODEL HAS {} PARAMETERS.'.format(model.count_params()))","metadata":{"execution":{"iopub.status.busy":"2021-12-28T13:29:42.054783Z","iopub.execute_input":"2021-12-28T13:29:42.05531Z","iopub.status.idle":"2021-12-28T13:29:42.256256Z","shell.execute_reply.started":"2021-12-28T13:29:42.055272Z","shell.execute_reply":"2021-12-28T13:29:42.252707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model and specify optimizer, loss and metric\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy',f1_score])","metadata":{"execution":{"iopub.status.busy":"2021-12-28T13:29:50.466979Z","iopub.execute_input":"2021-12-28T13:29:50.467219Z","iopub.status.idle":"2021-12-28T13:29:50.484293Z","shell.execute_reply.started":"2021-12-28T13:29:50.467191Z","shell.execute_reply":"2021-12-28T13:29:50.483617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add callbacks to reduce the learning rate if needed, early stopping, and checkpoint saving\ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_f1_score', \n                                                  patience=2, \n                                                  verbose=1, \n                                                  mode='max',\n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', \n                                              verbose=1,\n                                              patience=20),\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model.h5', \n                                                monitor='val_f1_score',\n                                                mode='max',\n                                                verbose=1,\n                                                save_best_only=True)]\n\ndef plot_history(history):\n    his=pd.DataFrame(history.history)\n    plt.subplots(1,2,figsize=(16,8))\n    \n    #loss:\n    plt.subplot(1,2,1)\n    plt.plot(range(len(his)),his['loss'],color='g',label='training')\n    plt.plot(range(len(his)),his['val_loss'],color='r',label='validation')\n    plt.legend()\n    plt.title('Loss')\n    \n    #accuracy\n    plt.subplot(1,2,2)\n    plt.plot(range(len(his)),his['accuracy'],color='g',label='training_acc')\n    plt.plot(range(len(his)),his['val_accuracy'],color='r',label='validation_acc')\n    \n#     #f1_score\n#     plt.plot(range(len(his)),his['f1_score'],color='steelblue',label='training_f1')\n#     plt.plot(range(len(his)),his['val_f1_score'],color='maroon',label='validation_f1')\n    \n    plt.legend()\n    plt.title('accuracy')\n    \n    plt.show()              ","metadata":{"execution":{"iopub.status.busy":"2021-12-28T13:29:52.010264Z","iopub.execute_input":"2021-12-28T13:29:52.010945Z","iopub.status.idle":"2021-12-28T13:29:52.023402Z","shell.execute_reply.started":"2021-12-28T13:29:52.01091Z","shell.execute_reply":"2021-12-28T13:29:52.022587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"his=model.fit(train_specs, \n          train_labels,\n          batch_size=32,\n          validation_split=0.2,\n          callbacks=callbacks,\n          epochs=EPOCHS)\n\nplot_history(his)","metadata":{"execution":{"iopub.status.busy":"2021-12-28T13:29:56.62774Z","iopub.execute_input":"2021-12-28T13:29:56.628124Z","iopub.status.idle":"2021-12-28T13:33:21.336834Z","shell.execute_reply.started":"2021-12-28T13:29:56.62809Z","shell.execute_reply":"2021-12-28T13:33:21.336212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}