{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"}],"dockerImageVersionId":30648,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install tensorflow_addons","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:16:13.748464Z","iopub.execute_input":"2024-02-25T15:16:13.748800Z","iopub.status.idle":"2024-02-25T15:16:28.241407Z","shell.execute_reply.started":"2024-02-25T15:16:13.748772Z","shell.execute_reply":"2024-02-25T15:16:28.240248Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nimport pandas as pd\nimport librosa\nimport numpy as np\nimport pickle\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow_addons.metrics import F1Score","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:16:28.243269Z","iopub.execute_input":"2024-02-25T15:16:28.243541Z","iopub.status.idle":"2024-02-25T15:16:43.063640Z","shell.execute_reply.started":"2024-02-25T15:16:28.243515Z","shell.execute_reply":"2024-02-25T15:16:43.062654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Global vars\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5 # seconds\nSPEC_SHAPE = (48, 128) # height x width\nFMIN = 500\nFMAX = 12500\n# MAX_AUDIO_FILES = 10000\nEPOCHS=50","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:16:43.065407Z","iopub.execute_input":"2024-02-25T15:16:43.066087Z","iopub.status.idle":"2024-02-25T15:16:43.071628Z","shell.execute_reply.started":"2024-02-25T15:16:43.066053Z","shell.execute_reply":"2024-02-25T15:16:43.070526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preparation","metadata":{}},{"cell_type":"code","source":"# Code adapted from: \n# https://www.kaggle.com/frlemarchand/bird-song-classification-using-an-efficientnet\n# Make sure to check out the entire notebook.\n\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2021/train_metadata.csv',)\n\n# Limit the number of training samples and classes\n# First, only use high quality samples\ntrain = train.query('rating>=4.8')\n\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), \n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 175] \n\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:16:43.072671Z","iopub.execute_input":"2024-02-25T15:16:43.072906Z","iopub.status.idle":"2024-02-25T15:16:43.584094Z","shell.execute_reply.started":"2024-02-25T15:16:43.072884Z","shell.execute_reply":"2024-02-25T15:16:43.583098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('LABELS.pkl','wb') as f:\n    pickle.dump(LABELS,f)","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:16:43.586117Z","iopub.execute_input":"2024-02-25T15:16:43.586375Z","iopub.status.idle":"2024-02-25T15:16:43.591120Z","shell.execute_reply.started":"2024-02-25T15:16:43.586353Z","shell.execute_reply":"2024-02-25T15:16:43.589931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## extracting spectrograms","metadata":{}},{"cell_type":"code","source":"# Shuffle the training data and limit the number of audio files to MAX_AUDIO_FILES\nTRAIN = shuffle(TRAIN, random_state=RANDOM_SEED)\n\n# Define a function that splits an audio file, \n# extracts spectrograms and saves them in a working directory\ndef get_spectrograms(filepath, primary_label, output_dir):\n    \n    # Open the file with librosa (limited to the first 15 seconds)\n    sig, rate = librosa.load(filepath, sr=SAMPLE_RATE, offset=None, duration=15)\n    \n    # Split signal into five second chunks\n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n        # End of signal?\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n        \n        sig_splits.append(split)\n        \n    # Extract mel spectrograms for each audio chunk\n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:\n        \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                  sr=SAMPLE_RATE, \n                                                  n_fft=1024, \n                                                  hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], \n                                                  fmin=FMIN, \n                                                  fmax=FMAX)\n    \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        \n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        # Save as image file\n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] + \n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n        \n        saved_samples.append(save_path)\n        s_cnt += 1\n        \n        \n    return saved_samples\n\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:16:47.276258Z","iopub.execute_input":"2024-02-25T15:16:47.277005Z","iopub.status.idle":"2024-02-25T15:16:47.290563Z","shell.execute_reply.started":"2024-02-25T15:16:47.276971Z","shell.execute_reply":"2024-02-25T15:16:47.289495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse audio files and extract training samples\ninput_dir = '../input/birdclef-2021/train_short_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\nsamples = []\nwith tqdm(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n        \n        if row.primary_label in most_represented_birds:\n            audio_file_path = os.path.join(input_dir, row.primary_label, row.filename)\n            samples += get_spectrograms(audio_file_path, row.primary_label, output_dir)\n            \nTRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\nprint('SUCCESSFULLY EXTRACTED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:18:57.379118Z","iopub.execute_input":"2024-02-25T15:18:57.379508Z","iopub.status.idle":"2024-02-25T15:20:07.958668Z","shell.execute_reply.started":"2024-02-25T15:18:57.379477Z","shell.execute_reply":"2024-02-25T15:20:07.957401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the first 12 spectrograms of TRAIN_SPECS\nplt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(TRAIN_SPECS[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(TRAIN_SPECS[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:20:07.964765Z","iopub.execute_input":"2024-02-25T15:20:07.965530Z","iopub.status.idle":"2024-02-25T15:20:10.176781Z","shell.execute_reply.started":"2024-02-25T15:20:07.965471Z","shell.execute_reply":"2024-02-25T15:20:10.175834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load training samples","metadata":{}},{"cell_type":"code","source":"# Parse all samples and add spectrograms into train data, primary_labels into label data\ntrain_specs, train_labels = [], []\nwith tqdm(total=len(TRAIN_SPECS)) as pbar:\n    for path in TRAIN_SPECS:\n        pbar.update(1)\n\n        # Open image\n        spec = Image.open(path)\n\n        # Convert to numpy array\n        spec = np.array(spec, dtype='float32')\n        \n        # Normalize between 0.0 and 1.0\n        # and exclude samples with nan \n        spec -= spec.min()\n        spec /= spec.max()\n        if not spec.max() == 1.0 or not spec.min() == 0.0:\n            continue\n\n        # Add channel axis to 2D array\n        spec = np.expand_dims(spec, -1)\n\n        # Add new dimension for batch size\n        spec = np.expand_dims(spec, 0)\n\n        # Add to train data\n        if len(train_specs) == 0:\n            train_specs = spec\n        else:\n            train_specs = np.vstack((train_specs, spec))\n\n        # Add to label data\n        target = np.zeros((len(LABELS)), dtype='float32')\n        bird = path.split(os.sep)[-2]\n        target[LABELS.index(bird)] = 1.0\n        if len(train_labels) == 0:\n            train_labels = target\n        else:\n            train_labels = np.vstack((train_labels, target))","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:20:10.178121Z","iopub.execute_input":"2024-02-25T15:20:10.178458Z","iopub.status.idle":"2024-02-25T15:20:55.364515Z","shell.execute_reply.started":"2024-02-25T15:20:10.178430Z","shell.execute_reply":"2024-02-25T15:20:55.363526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models","metadata":{}},{"cell_type":"code","source":"configurations = [\n    {\"filters\": 32, \"kernel_size\": (3, 3), \"stride\": (1, 1), \"pool_size\": (3, 3)},\n    {\"filters\": 32, \"kernel_size\": (3, 3), \"stride\": (1, 1), \"pool_size\": (2, 2)}, \n    {\"filters\": 32, \"kernel_size\": (5, 5), \"stride\": (1, 1), \"pool_size\": (2, 2)},\n    \n    {\"filters\": 64, \"kernel_size\": (3, 3), \"stride\": (1, 1), \"pool_size\": (2, 2)},\n    {\"filters\": 64, \"kernel_size\": (3, 3), \"stride\": (2, 2), \"pool_size\": (2, 2)},\n    \n]","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:54:34.790576Z","iopub.execute_input":"2024-02-25T15:54:34.791004Z","iopub.status.idle":"2024-02-25T15:54:34.797886Z","shell.execute_reply.started":"2024-02-25T15:54:34.790971Z","shell.execute_reply":"2024-02-25T15:54:34.796965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score=F1Score(num_classes=len(LABELS),average='macro',name='f1_score')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add callbacks to reduce the learning rate if needed, early stopping, and checkpoint saving\ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_f1_score', \n                                                  patience=2, \n                                                  verbose=1, \n                                                  mode='max',\n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', \n                                              verbose=1,\n                                              patience=20),\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model.h5', \n                                                monitor='val_f1_score',\n                                                mode='max',\n                                                verbose=1,\n                                                save_best_only=True)]","metadata":{"execution":{"iopub.status.busy":"2024-02-25T16:07:22.128444Z","iopub.execute_input":"2024-02-25T16:07:22.129173Z","iopub.status.idle":"2024-02-25T16:07:22.135540Z","shell.execute_reply.started":"2024-02-25T16:07:22.129133Z","shell.execute_reply":"2024-02-25T16:07:22.134487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize lists to store histories\ntrain_accuracy_histories = []\ntrain_loss_histories = []\nval_accuracy_histories = []\nval_loss_histories = []\n\n# Iterate over configurations\nfor config in configurations:\n    print(\"Testing configuration:\", config)\n    \n    \n    # Build the model\n    model = tf.keras.Sequential([\n        \n        # First conv block\n        tf.keras.layers.Conv2D(config[\"filters\"], config[\"kernel_size\"], strides=config[\"stride\"], activation='relu', \n                               input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1), padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(pool_size=config[\"pool_size\"]),\n        \n        # second conv block\n        tf.keras.layers.Conv2D(config[\"filters\"]*2, config[\"kernel_size\"], strides=config[\"stride\"], activation='relu', \n                               input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1), padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(pool_size=config[\"pool_size\"]),\n        \n        # Third conv block\n        tf.keras.layers.Conv2D(config[\"filters\"]*4, config[\"kernel_size\"], strides=config[\"stride\"], activation='relu', \n                               input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1), padding='same'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPooling2D(pool_size=config[\"pool_size\"]),\n        \n        \n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(256, activation='relu'),  \n        tf.keras.layers.Dropout(0.5),\n        tf.keras.layers.Dense(len(LABELS), activation='softmax')\n    ])\n    \n    print('MODEL HAS {} PARAMETERS.'.format(model.count_params()))\n    # Compile the model\n    model.compile(optimizer=tf.keras.optimizers.Adam(lr=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy',f1_score])\n    \n    # Train the model and store history\n    history = model.fit(train_specs, train_labels, epochs=30, batch_size=32, validation_split=0.2,callbacks=callbacks)\n    train_accuracy_histories.append(history.history['accuracy'])\n    train_loss_histories.append(history.history['loss'])\n    val_accuracy_histories.append(history.history['val_accuracy'])\n    val_loss_histories.append(history.history['val_loss'])","metadata":{"execution":{"iopub.status.busy":"2024-02-25T16:07:57.069591Z","iopub.execute_input":"2024-02-25T16:07:57.069999Z","iopub.status.idle":"2024-02-25T16:10:19.324620Z","shell.execute_reply.started":"2024-02-25T16:07:57.069969Z","shell.execute_reply":"2024-02-25T16:10:19.323608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Comparing","metadata":{}},{"cell_type":"code","source":"# Plot training accuracy history for each configuration\nplt.figure(figsize=(10, 5))\nfor i, train_acc in enumerate(train_accuracy_histories):\n    plt.plot(train_acc, label=f'Train Config {i+1}')\n\nplt.title('Training Accuracy History')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n\n# Plot validation accuracy history for each configuration\nplt.figure(figsize=(10, 5))\nfor i, val_acc in enumerate(val_accuracy_histories):\n    plt.plot(val_acc, label=f'Validation Config {i+1}')\n\nplt.title('Validation Accuracy History')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n\n# Plot training loss history for each configuration\nplt.figure(figsize=(10, 5))\nfor i, train_loss in enumerate(train_loss_histories):\n    plt.plot(train_loss, label=f'Train Config {i+1}')\n\nplt.title('Training Loss History')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\n# Plot validation loss history for each configuration\nplt.figure(figsize=(10, 5))\nfor i, val_loss in enumerate(val_loss_histories):\n    plt.plot(val_loss, label=f'Validation Config {i+1}')\n\nplt.title('Validation Loss History')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-25T16:10:37.828842Z","iopub.execute_input":"2024-02-25T16:10:37.829277Z","iopub.status.idle":"2024-02-25T16:10:38.914678Z","shell.execute_reply.started":"2024-02-25T16:10:37.829246Z","shell.execute_reply":"2024-02-25T16:10:38.913793Z"},"trusted":true},"execution_count":null,"outputs":[]}]}