{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":171110068,"sourceType":"kernelVersion"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-09T08:10:27.511864Z","iopub.execute_input":"2024-04-09T08:10:27.512271Z","iopub.status.idle":"2024-04-09T08:11:44.521942Z","shell.execute_reply.started":"2024-04-09T08:10:27.51224Z","shell.execute_reply":"2024-04-09T08:11:44.520764Z"}}},{"cell_type":"code","source":"import tensorflow as tf\nimport keras_cv\nimport numpy as np\nimport copy\nfrom joblib import load\nimport pandas as pd\nimport copy\n\nimport librosa.display\nimport matplotlib.pyplot as plt\n\nimport librosa\n\nfrom joblib import Parallel, delayed\n\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.preprocessing import LabelBinarizer\n\nimport os, gc","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:32.840216Z","iopub.execute_input":"2024-04-10T06:00:32.840533Z","iopub.status.idle":"2024-04-10T06:00:52.609555Z","shell.execute_reply.started":"2024-04-10T06:00:32.840506Z","shell.execute_reply":"2024-04-10T06:00:52.608777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Training data","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/birdclef-2024/sample_submission.csv')\ncolumns = list(sub.columns)[1:len(sub.columns)]","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:52.611168Z","iopub.execute_input":"2024-04-10T06:00:52.611669Z","iopub.status.idle":"2024-04-10T06:00:52.630381Z","shell.execute_reply.started":"2024-04-10T06:00:52.611645Z","shell.execute_reply":"2024-04-10T06:00:52.629499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = {}\ninverse_labels = {}\nfor i,j in enumerate(columns):\n    labels[j] = i\n    inverse_labels[i] = j","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:52.631647Z","iopub.execute_input":"2024-04-10T06:00:52.631993Z","iopub.status.idle":"2024-04-10T06:00:52.639921Z","shell.execute_reply.started":"2024-04-10T06:00:52.631964Z","shell.execute_reply":"2024-04-10T06:00:52.639008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\n\ntrain_path = '/kaggle/input/birdclef-2024/train_audio/'\nbirds = os.listdir(train_path)\nprint(f'number of birds: {len(birds)}')\n\naudio_count = 0\nfor i in tqdm(birds):\n    audios = os.listdir(train_path+i)\n    audio_count += len(audios)\nprint(f'Total audio files: {audio_count}')\n\ndef parse_array(x):\n    return ast.literal_eval(x)\n\nmeta = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv', converters={'secondary_labels': parse_array})\nprint(meta.shape)\n\nmeta = meta[meta.rating>2.5]\n\nmeta = meta.drop(columns = [ 'type', 'longitude', 'latitude', 'scientific_name', 'common_name', 'license', 'url'])","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:52.641817Z","iopub.execute_input":"2024-04-10T06:00:52.642082Z","iopub.status.idle":"2024-04-10T06:00:59.289972Z","shell.execute_reply.started":"2024-04-10T06:00:52.642060Z","shell.execute_reply":"2024-04-10T06:00:59.289070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_encoder = LabelEncoder()\nmeta['y'] = label_encoder.fit_transform(meta['primary_label'])\n\nlb = LabelBinarizer()\ny_oh =  lb.fit_transform(meta['primary_label'])","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:59.291483Z","iopub.execute_input":"2024-04-10T06:00:59.291865Z","iopub.status.idle":"2024-04-10T06:00:59.580255Z","shell.execute_reply.started":"2024-04-10T06:00:59.291833Z","shell.execute_reply":"2024-04-10T06:00:59.579415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_oh = []\n\nfor i in tqdm(meta.values):\n    temp = np.zeros(182)\n    temp[labels[i[0]]] = 1\n    for j in i[1]:\n        if j in labels:\n            temp[labels[j]] = .1\n    y_oh.append(temp)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:59.581412Z","iopub.execute_input":"2024-04-10T06:00:59.581719Z","iopub.status.idle":"2024-04-10T06:00:59.664212Z","shell.execute_reply.started":"2024-04-10T06:00:59.581692Z","shell.execute_reply":"2024-04-10T06:00:59.663342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(meta.primary_label.unique())","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:59.665501Z","iopub.execute_input":"2024-04-10T06:00:59.665808Z","iopub.status.idle":"2024-04-10T06:00:59.678148Z","shell.execute_reply.started":"2024-04-10T06:00:59.665783Z","shell.execute_reply":"2024-04-10T06:00:59.677206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = meta.drop(columns = ['author', 'rating', 'primary_label', 'secondary_labels'])\nprint(meta.shape)\nmeta = meta.reset_index()\nmeta = meta.drop(columns = ['index'])\nmeta.head()\ny_oh = np.asarray(y_oh)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:59.679297Z","iopub.execute_input":"2024-04-10T06:00:59.679627Z","iopub.status.idle":"2024-04-10T06:00:59.719457Z","shell.execute_reply.started":"2024-04-10T06:00:59.679597Z","shell.execute_reply":"2024-04-10T06:00:59.718230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataloader","metadata":{}},{"cell_type":"code","source":"def make_spectrograms(auds):\n    arr = []\n    \n    for row in (auds):\n        S = librosa.feature.melspectrogram(y=row, sr=32000, fmin = 100, fmax = 13000   )\n        arr.append(librosa.power_to_db(S, ref=np.max)/255)\n    return np.asarray(arr)\n\nclass TrainDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, x_data = meta.values, y_data = y_oh, batch_size=32, shuffle=True, mode= 'train'):\n        self.x_data = x_data\n        self.y_data = y_data\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.mode = mode\n        self.indices = np.arange(len(self.x_data))\n        if self.shuffle:\n            np.random.shuffle(self.indices)\n    \n    def __len__(self):\n        return int(np.ceil(len(self.x_data) / self.batch_size))\n    \n    def aug(self,x, y):\n        augmenter = keras_cv.layers.Augmenter(\n                [\n                    keras_cv.layers.MixUp(alpha = 0.05),\n                ],\n            )\n        \n        inputs = {\"images\": tf.convert_to_tensor(x, dtype=tf.float32), \"labels\": tf.convert_to_tensor(y, dtype=tf.float32)}\n        outputs = inputs\n        outputs = augmenter(outputs)\n        return outputs['images'], outputs['labels']\n    \n    def __getitem__(self, index):\n        batch_indices = self.indices[index * self.batch_size:(index + 1) * self.batch_size]\n        x_batch = self.getSpecs(batch_indices)/255.0\n        y_batch = self.y_data[batch_indices]\n        x_batch = x_batch.reshape(-1,128,313,1)\n        \n        if self.mode =='aug':\n            x, y = self.aug(x_batch, y_batch)\n            \n        return x_batch, y_batch\n    \n    def getSpecs(self, batch_indices):\n        \n        sr = 32000\n        train_path = '/kaggle/input/birdclef-2024/train_audio/'\n        \n        x = []\n        for i in batch_indices:\n            audio_file = train_path + self.x_data[i][0]\n            y, _ = librosa.load(audio_file, sr=None, )\n            \n            total_duration = len(y) / sr\n            while total_duration < 5:\n                y = np.concatenate([y, y])\n                total_duration = len(y) / sr \n            start_idx = int((total_duration / 2 - 2.5) * sr)\n            middle_5_seconds = y[start_idx:start_idx + 5 * sr]\n            x.append( middle_5_seconds )\n        \n        return make_spectrograms(x)\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indices)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:59.720817Z","iopub.execute_input":"2024-04-10T06:00:59.721117Z","iopub.status.idle":"2024-04-10T06:00:59.739613Z","shell.execute_reply.started":"2024-04-10T06:00:59.721090Z","shell.execute_reply":"2024-04-10T06:00:59.738635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = TrainDataGenerator(batch_size=32, shuffle = False)\n\nfor x_batch, y_batch in (train_gen):\n    print(x_batch.shape, y_batch.shape)\n\n    \n    fig, axs = plt.subplots(2, 4, figsize=(20, 10))\n\n    for i, image in enumerate(x_batch[0:8]):\n        ax = axs[i // 4, i % 4]\n        ax.imshow(image, aspect='auto', origin='lower', cmap='viridis')\n        ax.set_title(f'Image {i+1}')\n        ax.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n    break","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:00:59.743238Z","iopub.execute_input":"2024-04-10T06:00:59.743539Z","iopub.status.idle":"2024-04-10T06:01:14.800186Z","shell.execute_reply.started":"2024-04-10T06:00:59.743509Z","shell.execute_reply":"2024-04-10T06:01:14.798783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def build_model():\n    \n    inp = tf.keras.Input(shape=(128,313,1))\n    \n    base_model = tf.keras.applications.EfficientNetB0(\n        include_top=False,\n        weights='imagenet',\n    )\n\n    x = inp\n    x = tf.keras.layers.Concatenate(axis=3)([x,x,x])\n    \n    # OUTPUT\n    x = base_model(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dense(182,activation='softmax', dtype='float32')(x)\n        \n    # COMPILE MODEL\n    model = tf.keras.Model(inputs=inp, outputs=x)\n    opt = tf.keras.optimizers.Adam(learning_rate = 1e-3)\n    loss = tf.keras.losses.CategoricalCrossentropy()\n\n    model.compile(loss=loss, optimizer = opt, metrics=[tf.keras.metrics.AUC()]) \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:01:14.802029Z","iopub.execute_input":"2024-04-10T06:01:14.803115Z","iopub.status.idle":"2024-04-10T06:01:14.812952Z","shell.execute_reply.started":"2024-04-10T06:01:14.803079Z","shell.execute_reply":"2024-04-10T06:01:14.811972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training ","metadata":{}},{"cell_type":"code","source":"def lrfn(epoch):\n        return [1e-3,1e-4,1e-4][epoch]\n\nLR = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:01:14.814033Z","iopub.execute_input":"2024-04-10T06:01:14.814300Z","iopub.status.idle":"2024-04-10T06:01:14.842669Z","shell.execute_reply.started":"2024-04-10T06:01:14.814277Z","shell.execute_reply":"2024-04-10T06:01:14.841514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nfor fold, (train_index, val_index) in enumerate(kf.split(meta)):\n    \n    print(f\"Fold {fold+1} - Train: {len(train_index)}, Validation: {len(val_index)}\")\n    \n    train_gen = TrainDataGenerator(x_data = meta.iloc[train_index].values, \n                                   y_data = y_oh[train_index], batch_size=32,  )\n    \n    valid_gen = TrainDataGenerator(x_data = meta.iloc[val_index].values, y_data = y_oh[val_index], \n                                                       batch_size=64, shuffle = False, )\n    model = build_model()\n    model.fit(train_gen, epochs=3, validation_data= valid_gen, callbacks=[LR])\n    model.save_weights('model_weight{fold}.weights.h5')\n    break","metadata":{"execution":{"iopub.status.busy":"2024-04-10T06:01:14.843875Z","iopub.execute_input":"2024-04-10T06:01:14.844150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_gen = TrainDataGenerator(x_data = meta.iloc[val_index].values, y_data = y_oh[val_index], batch_size=64, mode ='valid')\nloss, accuracy = model.evaluate(valid_gen)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(loss, accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}