{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":6117,"sourceType":"modelInstanceVersion","modelInstanceId":4600},{"sourceId":6118,"sourceType":"modelInstanceVersion","modelInstanceId":4601},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport librosa\nfrom tqdm import tqdm\nfrom multiprocessing import Pool\nimport torch\nimport torchaudio\nimport tensorflow as tf\nimport tensorflow_io as tfio\nimport random\nimport tensorflow as tf\nimport keras_cv\nimport tensorflow_datasets as tfds\nimport keras","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:08:08.555747Z","iopub.execute_input":"2024-05-14T19:08:08.556054Z","iopub.status.idle":"2024-05-14T19:08:27.744308Z","shell.execute_reply.started":"2024-05-14T19:08:08.556028Z","shell.execute_reply":"2024-05-14T19:08:27.743540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    seed=42\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    num_classes = len(class_names)\n    labels = class_names\n    ids = [i for i in range(0, len(labels))]\n    label2id = {label: id for label, id in zip(labels, ids)}\n    id2label = {id: label for id, label in zip(ids, labels)}\n    training_duration = 15\n    testing_duration = 5\n    sample_rate = 32000\n    nnft = 2048\n    imgx = 128\n    imgy = 384\n    window = 2048\n    audio_len = training_duration*sample_rate\n    hop_length = audio_len // (imgy - 1)\n    epochs = 20\n    batch = 64\n    \ntf.keras.utils.set_random_seed(Config.seed)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:16:45.030867Z","iopub.execute_input":"2024-05-14T19:16:45.031230Z","iopub.status.idle":"2024-05-14T19:16:45.039686Z","shell.execute_reply.started":"2024-05-14T19:16:45.031201Z","shell.execute_reply":"2024-05-14T19:16:45.038818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:16:49.715719Z","iopub.execute_input":"2024-05-14T19:16:49.716086Z","iopub.status.idle":"2024-05-14T19:16:49.720421Z","shell.execute_reply.started":"2024-05-14T19:16:49.716059Z","shell.execute_reply":"2024-05-14T19:16:49.719509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_meta_data = pd.read_csv(BASE_PATH + '/train_metadata.csv')\ndata = audio_meta_data[['primary_label', 'filename']]\ndata['filepath'] = BASE_PATH + '/train_audio/' +data['filename']\ndata['label'] = data['primary_label'].map(Config.label2id)\ndata","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:16:49.909998Z","iopub.execute_input":"2024-05-14T19:16:49.910297Z","iopub.status.idle":"2024-05-14T19:16:50.043275Z","shell.execute_reply.started":"2024-05-14T19:16:49.910272Z","shell.execute_reply":"2024-05-14T19:16:50.042386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"\nplt.figure(figsize=(25, 5)) \ndata['primary_label'].value_counts().plot(kind='bar')  \nplt.xlabel('Primary Label')\nplt.ylabel('Frequency')\nplt.title('Frequency of Primary Labels')\nplt.xticks(rotation=60, fontsize='small')  \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:16:52.479656Z","iopub.execute_input":"2024-05-14T19:16:52.479999Z","iopub.status.idle":"2024-05-14T19:16:53.910765Z","shell.execute_reply.started":"2024-05-14T19:16:52.479974Z","shell.execute_reply":"2024-05-14T19:16:53.909769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_amplitude(audio):\n    plt.figure(figsize=(10, 4))\n    librosa.display.waveshow(audio[:Config.training_duration*Config.sample_rate], sr=Config.sample_rate)\n    plt.title('Waveform of Audio')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Amplitude')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:16:54.142572Z","iopub.execute_input":"2024-05-14T19:16:54.142949Z","iopub.status.idle":"2024-05-14T19:16:54.148865Z","shell.execute_reply.started":"2024-05-14T19:16:54.142919Z","shell.execute_reply":"2024-05-14T19:16:54.147907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath =  data['filepath'][0]\nprint(\"data for this file--->\", filepath)\naudio, sample_rate = librosa.load(filepath, sr = None)\nprint(audio, sample_rate)\nprint(len(audio))\nplot_amplitude(audio)\nlibrosa.get_duration(y=audio, sr=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:14:06.468724Z","iopub.execute_input":"2024-05-14T19:14:06.469099Z","iopub.status.idle":"2024-05-14T19:14:15.608224Z","shell.execute_reply.started":"2024-05-14T19:14:06.469072Z","shell.execute_reply":"2024-05-14T19:14:15.607346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# def process_file(filepath):\n#     audio, sample_rate = librosa.load(filepath, sr=None)\n#     duration = librosa.get_duration(y=audio, sr=sample_rate)\n#     return duration < training_duration\n\n# num_processes = 5\n# with Pool(num_processes) as pool:\n#     results = list(tqdm(pool.imap(process_file, data['filepath']), total=len(data['filepath'])))\n# cnt = sum(results)\n# print(cnt)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:00:52.661543Z","iopub.execute_input":"2024-05-12T10:00:52.662242Z","iopub.status.idle":"2024-05-12T10:00:52.667061Z","shell.execute_reply.started":"2024-05-12T10:00:52.662205Z","shell.execute_reply":"2024-05-12T10:00:52.665932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Pre-Processing\n","metadata":{}},{"cell_type":"code","source":"class DataPreProcessor:\n    def __init__(self, sample_rate, training_duration):\n        self.sample_rate = sample_rate\n        self.training_duration = training_duration\n\n    def load_audio(self, filepath, trim=False, fade=False):\n        audio = tfio.audio.AudioIOTensor(filepath)\n        audio = audio[:]\n        if trim:\n            position = tfio.audio.trim(audio, axis=0, epsilon=0.1)\n            start = position[0]\n            stop = position[1]\n            audio = audio[start[0]:stop[0]]\n        if fade:\n            audio = tfio.audio.fade(audio, fade_in=1000, fade_out=2000, mode=\"logarithmic\")\n        audio = tf.squeeze(audio, axis=-1)\n        return audio\n\n    def resize_audio(self, audio, target_len):\n        audio_len = tf.shape(audio)[0]\n        diff_len = abs(target_len - audio_len)\n        if audio_len < target_len:\n            pad1 = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            pad2 = diff_len - pad1\n            audio = tf.pad(audio, paddings=[[pad1, pad2]], mode=\"constant\")\n        elif audio_len > target_len:\n            idx = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            audio = audio[idx : (idx + target_len)]\n        return tf.reshape(audio, [target_len])\n\n    def get_spectrogram(self, audio):\n        spec = tf.keras.layers.MelSpectrogram(\n            fft_length=Config.nnft,\n            sequence_stride=Config.hop_length,\n            sampling_rate=self.sample_rate,\n            num_mel_bins=Config.imgx)(audio)\n        return spec\n\n    def process_spectrogram(self, spec):\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        min_val = tf.reduce_min(spec)\n        max_val = tf.reduce_max(spec)\n        spec = tf.where(tf.math.equal(max_val - min_val, 0), spec - min_val, (spec - min_val) / (max_val - min_val))\n        return spec\n\n    def process_file(self, filepath):\n        audio = self.load_audio(filepath)\n        audio = self.resize_audio(audio, self.training_duration * self.sample_rate)\n        spec = self.get_spectrogram(audio)\n        spec= self.process_spectrogram(spec)\n        spec = tf.tile(spec[..., None], [1, 1, 3])\n        return spec\n\nprocessor = DataPreProcessor(Config.sample_rate, Config.training_duration)\n\n# Process a file\nspec = processor.process_file(data['filepath'][1])\n\nplt.figure(figsize=(10, 4))\nplt.imshow(tf.squeeze(spec).numpy(), aspect='auto', cmap='viridis')\nplt.title('Spectrogram')\nplt.xlabel('Time')\nplt.ylabel('Frequency')\nplt.colorbar(label='Magnitude')\nplt.show()\nprint(spec)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.drop(columns = 'primary_label')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:28.290340Z","iopub.execute_input":"2024-05-14T19:18:28.290698Z","iopub.status.idle":"2024-05-14T19:18:28.298275Z","shell.execute_reply.started":"2024-05-14T19:18:28.290669Z","shell.execute_reply":"2024-05-14T19:18:28.297354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data= data.drop(columns = ['filename'])","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:28.596294Z","iopub.execute_input":"2024-05-14T19:18:28.596603Z","iopub.status.idle":"2024-05-14T19:18:28.605568Z","shell.execute_reply.started":"2024-05-14T19:18:28.596578Z","shell.execute_reply":"2024-05-14T19:18:28.604593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:29.581610Z","iopub.execute_input":"2024-05-14T19:18:29.582228Z","iopub.status.idle":"2024-05-14T19:18:29.593147Z","shell.execute_reply.started":"2024-05-14T19:18:29.582198Z","shell.execute_reply":"2024-05-14T19:18:29.592035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_df, valid_df = train_test_split(data, test_size=0.2)\nprint(f\"Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:30.100434Z","iopub.execute_input":"2024-05-14T19:18:30.101251Z","iopub.status.idle":"2024-05-14T19:18:30.231531Z","shell.execute_reply.started":"2024-05-14T19:18:30.101218Z","shell.execute_reply":"2024-05-14T19:18:30.230610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:30.503274Z","iopub.execute_input":"2024-05-14T19:18:30.503590Z","iopub.status.idle":"2024-05-14T19:18:30.508684Z","shell.execute_reply.started":"2024-05-14T19:18:30.503564Z","shell.execute_reply":"2024-05-14T19:18:30.507826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:30.681767Z","iopub.execute_input":"2024-05-14T19:18:30.682228Z","iopub.status.idle":"2024-05-14T19:18:30.688212Z","shell.execute_reply.started":"2024-05-14T19:18:30.682191Z","shell.execute_reply":"2024-05-14T19:18:30.687306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:31.045578Z","iopub.execute_input":"2024-05-14T19:18:31.045955Z","iopub.status.idle":"2024-05-14T19:18:31.057399Z","shell.execute_reply.started":"2024-05-14T19:18:31.045926Z","shell.execute_reply":"2024-05-14T19:18:31.056357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_augmenter():\n    augmenters = [\n        keras_cv.layers.MixUp(alpha=0.4),\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                     width_factor=(0.06, 0.12)), # time-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                     width_factor=(1.0, 1.0)), # freq-masking\n    ]\n    \n    def augment(img, label):\n        data = {\"images\":img, \"labels\":label}\n        for augmenter in augmenters:\n            if tf.random.uniform([]) < 0.35:\n                data = augmenter(data, training=True)\n        return data[\"images\"], data[\"labels\"]\n    \n    return augment","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:31.393575Z","iopub.execute_input":"2024-05-14T19:18:31.394393Z","iopub.status.idle":"2024-05-14T19:18:31.401163Z","shell.execute_reply.started":"2024-05-14T19:18:31.394362Z","shell.execute_reply":"2024-05-14T19:18:31.400108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(filepaths, labels, sample_rate=Config.sample_rate, training_duration=Config.training_duration, batch_size=Config.batch):\n    processor = DataPreProcessor(sample_rate, training_duration)\n    \n    def generator():\n        for filepath, label in zip(filepaths, labels):\n            spectrogram = processor.process_file(filepath)\n            yield spectrogram, label\n            \n    augment_fn = build_augmenter()\n    \n    dataset = tf.data.Dataset.from_generator(generator, output_signature=(\n        tf.TensorSpec(shape=processor.process_file(filepaths[0]).shape, dtype=tf.float32), \n        tf.TensorSpec(shape=(182,), dtype=tf.float32)))  # Assuming labels are one-hot encoded\n    \n    \n    \n    # Cache the preprocessed data\n    dataset = dataset.cache()\n    \n    # Batch the dataset\n    dataset = dataset.batch(batch_size, drop_remainder=True)\n    \n    dataset = dataset.map(augment_fn, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    \n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    \n    \n    \n    return dataset\n","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:31.577947Z","iopub.execute_input":"2024-05-14T19:18:31.578235Z","iopub.status.idle":"2024-05-14T19:18:31.586030Z","shell.execute_reply.started":"2024-05-14T19:18:31.578211Z","shell.execute_reply":"2024-05-14T19:18:31.585046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths = train_df.filepath.values\ntrain_labels = train_df.label.values\nvalid_paths = valid_df.filepath.values\nvalid_labels = valid_df.label.values\nnum_classes = 182\ntrain_labels_one_hot = tf.one_hot(train_labels, num_classes)\nvalid_labels_one_hot = tf.one_hot(valid_labels, num_classes)\ntrain_dataset = create_dataset(filepaths = train_paths, labels = train_labels_one_hot)\nvalid_dataset = create_dataset(filepaths = valid_paths, labels = valid_labels_one_hot)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:18:31.754813Z","iopub.execute_input":"2024-05-14T19:18:31.755104Z","iopub.status.idle":"2024-05-14T19:18:33.900910Z","shell.execute_reply.started":"2024-05-14T19:18:31.755080Z","shell.execute_reply":"2024-05-14T19:18:33.900072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inp = keras.layers.Input(shape=(None, None, 3))\n# Pretrained backbone\nbackbone = keras_cv.models.EfficientNetV2Backbone.from_preset(\n    'efficientnetv2_b2_imagenet',\n)\nout = keras_cv.models.ImageClassifier(\n    backbone=backbone,\n    num_classes=182,\n    name=\"classifier\"\n)(inp)\n# Build model\nmodel = keras.models.Model(inputs=inp, outputs=out)\n# Compile model with optimizer, loss and metrics\nmodel.compile(optimizer=\"adam\",\n              loss=keras.losses.CategoricalCrossentropy(label_smoothing=0.02),\n              metrics=[keras.metrics.AUC(name='auc')],\n             )\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:19:15.866001Z","iopub.execute_input":"2024-05-14T19:19:15.866739Z","iopub.status.idle":"2024-05-14T19:19:21.190206Z","shell.execute_reply.started":"2024-05-14T19:19:15.866710Z","shell.execute_reply":"2024-05-14T19:19:21.189285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\ndef get_lr_callback(batch_size=64, mode='cos', epochs=Config.epochs, plot=False):\n    lr_start, lr_max, lr_min = 5e-5, 8e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        elif mode == 'exp': lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n        elif mode == 'step': lr = lr_max * lr_decay**((epoch - lr_ramp_ep - lr_sus_ep) // 2)\n        elif mode == 'cos':\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        return lr\n\n    if plot:  # Plot lr curve if plot is True\n        plt.figure(figsize=(10, 5))\n        plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('lr')\n        plt.title('LR Scheduler')\n        plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  # Create lr callback\nlr_cb = get_lr_callback(Config.batch, plot=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:19:21.191738Z","iopub.execute_input":"2024-05-14T19:19:21.192031Z","iopub.status.idle":"2024-05-14T19:19:21.460249Z","shell.execute_reply.started":"2024-05-14T19:19:21.192007Z","shell.execute_reply":"2024-05-14T19:19:21.459333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.weights.h5\",\n                                         monitor='val_auc',\n                                         save_best_only=True,\n                                         save_weights_only=True,\n                                         mode='max')","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:19:21.461433Z","iopub.execute_input":"2024-05-14T19:19:21.462062Z","iopub.status.idle":"2024-05-14T19:19:21.466937Z","shell.execute_reply.started":"2024-05-14T19:19:21.462029Z","shell.execute_reply":"2024-05-14T19:19:21.466081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_dataset, \n    validation_data=valid_dataset, \n    epochs=Config.epochs,\n    callbacks=[lr_cb, ckpt_cb], \n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T19:19:21.468999Z","iopub.execute_input":"2024-05-14T19:19:21.469515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}