{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nimport pandas as pd\nimport librosa\nimport numpy as np\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\n\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5\nSPEC_SHAPE = (48, 128)\nFMIN = 500\nFMAX = 12500\nMAX_AUDIO_FILES = 1500","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-25T10:08:38.127360Z","iopub.execute_input":"2024-10-25T10:08:38.127759Z","iopub.status.idle":"2024-10-25T10:08:54.658149Z","shell.execute_reply.started":"2024-10-25T10:08:38.127719Z","shell.execute_reply":"2024-10-25T10:08:54.657003Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('../input/birdclef-2021/train_metadata.csv',)\n\ntrain = train.query('rating>=4')\n\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\n\nmost_represented_birds = [key for key,value in birds_count.items() if value >= 200] \n\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:08:54.659986Z","iopub.execute_input":"2024-10-25T10:08:54.660572Z","iopub.status.idle":"2024-10-25T10:08:55.200436Z","shell.execute_reply.started":"2024-10-25T10:08:54.660532Z","shell.execute_reply":"2024-10-25T10:08:55.199317Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN = shuffle(TRAIN, random_state=RANDOM_SEED)[:MAX_AUDIO_FILES]\n\ndef get_spectrograms(filepath, primary_label, output_dir):\n    sig, rate = librosa.load(filepath, sr=SAMPLE_RATE, offset=None, duration=15)\n    \n    sig_splits = []\n    for i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n        split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n        if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n            break\n        \n        sig_splits.append(split)\n        \n    s_cnt = 0\n    saved_samples = []\n    for chunk in sig_splits:        \n        hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n        mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                                  sr=SAMPLE_RATE, \n                                                  n_fft=1024, \n                                                  hop_length=hop_length, \n                                                  n_mels=SPEC_SHAPE[0], \n                                                  fmin=FMIN, \n                                                  fmax=FMAX)\n    \n        mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n        \n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        save_dir = os.path.join(output_dir, primary_label)\n        if not os.path.exists(save_dir):\n            os.makedirs(save_dir)\n        save_path = os.path.join(save_dir, filepath.rsplit(os.sep, 1)[-1].rsplit('.', 1)[0] + \n                                 '_' + str(s_cnt) + '.png')\n        im = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        im.save(save_path)\n        \n        saved_samples.append(save_path)\n        s_cnt += 1\n        \n        \n    return saved_samples\n\nprint('FINAL NUMBER OF AUDIO FILES IN TRAINING DATA:', len(TRAIN))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:08:55.201640Z","iopub.execute_input":"2024-10-25T10:08:55.202010Z","iopub.status.idle":"2024-10-25T10:08:55.220598Z","shell.execute_reply.started":"2024-10-25T10:08:55.201969Z","shell.execute_reply":"2024-10-25T10:08:55.219417Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_dir = '../input/birdclef-2021/train_short_audio/'\noutput_dir = '../working/melspectrogram_dataset/'\nsamples = []\nwith tqdm(total=len(TRAIN)) as pbar:\n    for idx, row in TRAIN.iterrows():\n        pbar.update(1)\n        \n        if row.primary_label in most_represented_birds:\n            audio_file_path = os.path.join(input_dir, row.primary_label, row.filename)\n            samples += get_spectrograms(audio_file_path, row.primary_label, output_dir)\n            \nTRAIN_SPECS = shuffle(samples, random_state=RANDOM_SEED)\nprint('SUCCESSFULLY EXTRACTED {} SPECTROGRAMS'.format(len(TRAIN_SPECS)))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:08:55.223179Z","iopub.execute_input":"2024-10-25T10:08:55.223551Z","iopub.status.idle":"2024-10-25T10:11:09.563490Z","shell.execute_reply.started":"2024-10-25T10:08:55.223500Z","shell.execute_reply":"2024-10-25T10:11:09.562189Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nfor i in range(12):\n    spec = Image.open(TRAIN_SPECS[i])\n    plt.subplot(3, 4, i + 1)\n    plt.title(TRAIN_SPECS[i].split(os.sep)[-1])\n    plt.imshow(spec, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:11:09.569377Z","iopub.execute_input":"2024-10-25T10:11:09.570791Z","iopub.status.idle":"2024-10-25T10:11:12.131945Z","shell.execute_reply.started":"2024-10-25T10:11:09.570716Z","shell.execute_reply":"2024-10-25T10:11:12.130965Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_specs, train_labels = [], []\nwith tqdm(total=len(TRAIN_SPECS)) as pbar:\n    for path in TRAIN_SPECS:\n        pbar.update(1)\n\n        spec = Image.open(path)\n\n        spec = np.array(spec, dtype='float32')\n        \n        spec -= spec.min()\n        spec /= spec.max()\n        if not spec.max() == 1.0 or not spec.min() == 0.0:\n            continue\n\n        spec = np.expand_dims(spec, -1)\n\n        spec = np.expand_dims(spec, 0)\n\n        if len(train_specs) == 0:\n            train_specs = spec\n        else:\n            train_specs = np.vstack((train_specs, spec))\n\n        target = np.zeros((len(LABELS)), dtype='float32')\n        bird = path.split(os.sep)[-2]\n        target[LABELS.index(bird)] = 1.0\n        if len(train_labels) == 0:\n            train_labels = target\n        else:\n            train_labels = np.vstack((train_labels, target))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:11:12.133300Z","iopub.execute_input":"2024-10-25T10:11:12.134010Z","iopub.status.idle":"2024-10-25T10:12:35.648416Z","shell.execute_reply.started":"2024-10-25T10:11:12.133960Z","shell.execute_reply":"2024-10-25T10:12:35.647316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def residual_block(x, filters):\n    shortcut = x\n    x = tf.keras.layers.Conv2D(filters, (3, 3), padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.LeakyReLU(alpha=0.1)(x)\n    \n    x = tf.keras.layers.Conv2D(filters, (3, 3), padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    if shortcut.shape[-1] != filters:\n        shortcut = tf.keras.layers.Conv2D(filters, (1, 1), padding='same')(shortcut)\n\n    x = tf.keras.layers.Add()([x, shortcut])\n    x = tf.keras.layers.LeakyReLU(alpha=0.1)(x)\n    \n    return x\n\ninput_shape = (SPEC_SHAPE[0], SPEC_SHAPE[1], 1)\n\ninputs = tf.keras.Input(shape=input_shape)\n\nx = tf.keras.layers.Conv2D(32, (3, 3), padding='same')(inputs)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.LeakyReLU(alpha=0.1)(x)\nx = tf.keras.layers.MaxPooling2D((2, 2))(x)\n\nx = residual_block(x, 64)\nx = tf.keras.layers.MaxPooling2D((2, 2))(x)\n\nx = residual_block(x, 128)\nx = tf.keras.layers.MaxPooling2D((2, 2))(x)\n\nx = residual_block(x, 256)\nx = tf.keras.layers.MaxPooling2D((2, 2))(x)\n\nx = residual_block(x, 512)\nx = tf.keras.layers.MaxPooling2D((2, 2))(x)\n\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\n\nx = tf.keras.layers.Dense(512, kernel_regularizer=tf.keras.regularizers.l2(0.001))(x)\nx = tf.keras.layers.LeakyReLU(alpha=0.1)(x)\nx = tf.keras.layers.Dropout(0.5)(x)\n\nx = tf.keras.layers.Dense(256, kernel_regularizer=tf.keras.regularizers.l2(0.001))(x)\nx = tf.keras.layers.LeakyReLU(alpha=0.1)(x)\nx = tf.keras.layers.Dropout(0.5)(x)\n\noutputs = tf.keras.layers.Dense(len(LABELS), activation='softmax')(x)\n\nmodel = tf.keras.Model(inputs=inputs, outputs=outputs)\n\nprint('MODEL HAS {} PARAMETERS.'.format(model.count_params()))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:47:35.688423Z","iopub.execute_input":"2024-10-25T10:47:35.688857Z","iopub.status.idle":"2024-10-25T10:47:35.949851Z","shell.execute_reply.started":"2024-10-25T10:47:35.688817Z","shell.execute_reply":"2024-10-25T10:47:35.948789Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:47:39.901876Z","iopub.execute_input":"2024-10-25T10:47:39.902628Z","iopub.status.idle":"2024-10-25T10:47:39.912807Z","shell.execute_reply.started":"2024-10-25T10:47:39.902577Z","shell.execute_reply":"2024-10-25T10:47:39.911789Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', \n                                                  patience=5, \n                                                  verbose=1, \n                                                  factor=0.5),\n             tf.keras.callbacks.EarlyStopping(monitor='val_accuracy', \n                                              verbose=1,\n                                              patience=10),\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model.keras', \n                                                monitor='val_loss',\n                                                verbose=0,\n                                                save_best_only=True)]","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:47:42.485609Z","iopub.execute_input":"2024-10-25T10:47:42.486308Z","iopub.status.idle":"2024-10-25T10:47:42.492881Z","shell.execute_reply.started":"2024-10-25T10:47:42.486265Z","shell.execute_reply":"2024-10-25T10:47:42.491802Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(train_specs, \n          train_labels,\n          batch_size=16,\n          validation_split=0.2,\n          callbacks=callbacks,\n          epochs=50)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:47:48.062703Z","iopub.execute_input":"2024-10-25T10:47:48.063154Z","iopub.status.idle":"2024-10-25T10:49:57.043827Z","shell.execute_reply.started":"2024-10-25T10:47:48.063111Z","shell.execute_reply":"2024-10-25T10:49:57.042656Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = tf.keras.models.load_model('best_model.keras')\n\nsoundscape_path = '../input/birdclef-2021/train_soundscapes/10534_SSW_20170429.ogg'\n\nsig, rate = librosa.load(soundscape_path, sr=SAMPLE_RATE)\n\ndata = {'row_id': [], 'prediction': [], 'score': []}\n\nsig_splits = []\nfor i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n    split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n    if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n        break\n\n    sig_splits.append(split)\n    \nseconds, scnt = 0, 0\nfor chunk in sig_splits:\n    seconds += 5\n        \n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=1024, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    \n    mel_spec = np.expand_dims(mel_spec, -1)\n\n    mel_spec = np.expand_dims(mel_spec, 0)\n    \n    p = model.predict(mel_spec)[0]\n    \n    idx = p.argmax()\n    species = LABELS[idx]\n    score = p[idx]\n    \n    data['row_id'].append(soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                          '_' + str(seconds))    \n    \n    if score > 0.3:\n        data['prediction'].append(species)\n        scnt += 1\n    else:\n        data['prediction'].append('nocall')\n        \n    data['score'].append(score)\n        \nprint('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:52:36.586530Z","iopub.execute_input":"2024-10-25T10:52:36.587461Z","iopub.status.idle":"2024-10-25T10:52:55.880955Z","shell.execute_reply.started":"2024-10-25T10:52:36.587416Z","shell.execute_reply":"2024-10-25T10:52:55.879587Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = pd.DataFrame(data, columns = ['row_id', 'prediction', 'score'])\n\ngt = pd.read_csv('../input/birdclef-2021/train_soundscape_labels.csv',)\nresults = pd.merge(gt, results, on='row_id')\n\nresults['birds_set'] = results['birds'].apply(lambda x: set(x.split(',')))  \nresults['prediction_set'] = results['prediction'].apply(lambda x: set(x.split(',')))  \n\nresults['correct'] = results.apply(lambda row: not row['birds_set'].isdisjoint(row['prediction_set']), axis=1)\n\naccuracy = results['correct'].mean()  \naccuracy_percentage = accuracy * 100  \n\nprint(f'Accuracy: {accuracy_percentage:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2024-10-25T10:54:16.959710Z","iopub.execute_input":"2024-10-25T10:54:16.960681Z","iopub.status.idle":"2024-10-25T10:54:16.989065Z","shell.execute_reply.started":"2024-10-25T10:54:16.960629Z","shell.execute_reply":"2024-10-25T10:54:16.987838Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}