{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"markdown","source":"#### This notebook aims at direct convertion of spectrogram from audio and train the spectrogram without saving the spectrogram images","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport librosa\nimport random\nfrom glob import glob\nfrom tqdm import tqdm\nimport sklearn\nimport librosa.display as lid\nimport IPython.display as ipd\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:44:39.518635Z","iopub.execute_input":"2023-03-20T14:44:39.519225Z","iopub.status.idle":"2023-03-20T14:44:48.265476Z","shell.execute_reply.started":"2023-03-20T14:44:39.519189Z","shell.execute_reply":"2023-03-20T14:44:48.264373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gpus = tf.config.experimental.list_physical_devices('GPU')\nif gpus:\n    try:\n        tf.config.experimental.set_visible_devices(gpus[0], 'GPU')\n        tf.config.experimental.set_memory_growth(gpus[0], True)\n    except RuntimeError as e:\n        print(e)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:44:54.898338Z","iopub.execute_input":"2023-03-20T14:44:54.899825Z","iopub.status.idle":"2023-03-20T14:44:55.12903Z","shell.execute_reply.started":"2023-03-20T14:44:54.899782Z","shell.execute_reply":"2023-03-20T14:44:55.127884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"class CFG:\n    base_path = '/kaggle/input/birdclef-2023/'\n    train_df = base_path + 'train_metadata.csv'\n    train_audio_folder = base_path + 'train_audio/'\n    submission_df = base_path + 'sample_submission.csv'\n    test_sound_path = base_path + 'test_soundscapes/'\n\n    model_type = 'baseline'\n    model_save_path = f'/kaggle/working/{model_type}.h5'\n    \n    seed = 1997\n    \n    batch_size = 64\n    learning_rate = 0.00003\n    img_size = 224\n    input_shape = (img_size,img_size, 3)\n    \n    normalize = True\n    \n    label_smoothing = 0.05\n    \n    num_classes = 264\n    epochs = 100\n    duration = 10\n    sampling_rate = 32000\n    audio_length = duration * sampling_rate\n    nfft = 2028\n    n_mels = 128\n    window = 2048\n    hop_length = 512\n    fmin = 20\n    fmax = 16000\n    normalize = True\n    \n    \n    freq_mask = 20\n    time_mask = 30\n    \n    timeshift_prob = 0.0\n    gn_prob = 0.35","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:45:00.818934Z","iopub.execute_input":"2023-03-20T14:45:00.820025Z","iopub.status.idle":"2023-03-20T14:45:00.82841Z","shell.execute_reply.started":"2023-03-20T14:45:00.819961Z","shell.execute_reply":"2023-03-20T14:45:00.827214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Filter Data","metadata":{}},{"cell_type":"code","source":"def upsample_data(df, thr=20):\n    # get the class distribution\n    class_dist = df['primary_label'].value_counts()\n\n    # identify the classes that have less than the threshold number of samples\n    down_classes = class_dist[class_dist < thr].index.tolist()\n\n    # create an empty list to store the upsampled dataframes\n    up_dfs = []\n\n    # loop through the undersampled classes and upsample them\n    for c in down_classes:\n        # get the dataframe for the current class\n        class_df = df.query(\"primary_label==@c\")\n        # find number of samples to add\n        num_up = thr - class_df.shape[0]\n        # upsample the dataframe\n        class_df = class_df.sample(n=num_up, replace=True, random_state=CFG.seed)\n        # append the upsampled dataframe to the list\n        up_dfs.append(class_df)\n\n    # concatenate the upsampled dataframes and the original dataframe\n    up_df = pd.concat([df] + up_dfs, axis=0, ignore_index=True)\n    \n    return up_df","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:47:13.48326Z","iopub.execute_input":"2023-03-20T14:47:13.484335Z","iopub.status.idle":"2023-03-20T14:47:13.492322Z","shell.execute_reply.started":"2023-03-20T14:47:13.484292Z","shell.execute_reply":"2023-03-20T14:47:13.491102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df = pd.read_csv(CFG.train_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:47:16.594397Z","iopub.execute_input":"2023-03-20T14:47:16.595124Z","iopub.status.idle":"2023-03-20T14:47:16.741578Z","shell.execute_reply.started":"2023-03-20T14:47:16.595084Z","shell.execute_reply":"2023-03-20T14:47:16.74053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Upsample data\nup_df = upsample_data(data_df, thr=100)\nprint(f'Before Upsample Size: {len(data_df)}')\nprint(f'After Upsample Size: {len(up_df)}')\n\n# # Show effect of upsample\n# fig, ax = plt.subplots(1, 1, figsize=(12, 6))\n# up_df.primary_label.value_counts()[:].plot.bar(ax=ax, color='green', label='w/ upsample')\n# data_df.primary_label.value_counts()[:].plot.bar(ax=ax, color='red', label='w/o upsample')\n# plt.xticks([])\n# plt.axhline(y=50, color='k', linestyle='--', label='threshold')\n# plt.legend()\n# plt.title(\"Effect of Upsample\")\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:47:32.886711Z","iopub.execute_input":"2023-03-20T14:47:32.887326Z","iopub.status.idle":"2023-03-20T14:47:33.499172Z","shell.execute_reply.started":"2023-03-20T14:47:32.887286Z","shell.execute_reply":"2023-03-20T14:47:33.497381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain, val = train_test_split(up_df, test_size=0.1, random_state=1997, stratify=up_df[['primary_label']])\ntrain.to_csv('train_df.csv', index = False)\nval.to_csv('val.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:48:47.043447Z","iopub.execute_input":"2023-03-20T14:48:47.043834Z","iopub.status.idle":"2023-03-20T14:48:47.422114Z","shell.execute_reply.started":"2023-03-20T14:48:47.043799Z","shell.execute_reply":"2023-03-20T14:48:47.421047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility","metadata":{}},{"cell_type":"code","source":"# Generates random integer\ndef random_int(shape=[], minval=0, maxval=1):\n    return tf.random.uniform(shape=shape, minval=minval, maxval=maxval, dtype=tf.int32)\n\n# Generats random float\ndef random_float(shape=[], minval=0.0, maxval=1.0):\n    rnd = tf.random.uniform(shape=shape, minval=minval, maxval=maxval, dtype=tf.float32)\n    return rnd","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:49:21.884502Z","iopub.execute_input":"2023-03-20T14:49:21.884956Z","iopub.status.idle":"2023-03-20T14:49:21.893234Z","shell.execute_reply.started":"2023-03-20T14:49:21.884917Z","shell.execute_reply":"2023-03-20T14:49:21.891988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Crop and Pad audio and Audio Augment","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef CropOrPad(audio, target_len, pad_mode='constant'):\n    # Get the length of the input audio\n    audio_len = tf.shape(audio)[0]\n    # If the length of the input audio is smaller than the target length, randomly pad the audio\n    if audio_len < target_len:\n        # Calculate the offset between the input audio and the target length\n        diff_len = (target_len - audio_len)\n        # Select a random location for padding\n        pad1 = random_int([], minval=0, maxval=diff_len)\n        # Calculate the second padding value\n        pad2 = diff_len - pad1\n        pad_len = [pad1, pad2]\n        # Apply padding to the audio data\n        audio = tf.pad(audio, paddings=[pad_len], mode=pad_mode)\n    # If the length of the input audio is larger than the target length, crop the audio\n    elif audio_len > target_len:\n        # Calculate the difference in length between the input audio and the target length\n        diff_len = (audio_len - target_len)\n        # Select a random location for cropping\n        idx = tf.random.uniform([], 0, diff_len, dtype=tf.int32)\n        # Crop the audio data\n        audio = audio[idx: (idx + target_len)]\n    # Reshape the audio data to the target length\n    audio = tf.reshape(audio, [target_len])\n    # Return the cropped or padded audio data\n    return audio\n\n@tf.function\ndef Normalize(data, min_max=True):\n    # Compute the mean and standard deviation of the data\n    MEAN = tf.math.reduce_mean(data)\n    STD = tf.math.reduce_std(data)\n    # Standardize the data\n    data = tf.math.divide_no_nan(data - MEAN, STD)\n    # Normalize to [0, 1]\n    if min_max:\n        MIN = tf.math.reduce_min(data)\n        MAX = tf.math.reduce_max(data)\n        data = tf.math.divide_no_nan(data - MIN, MAX - MIN)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:51:22.243475Z","iopub.execute_input":"2023-03-20T14:51:22.243922Z","iopub.status.idle":"2023-03-20T14:51:22.258764Z","shell.execute_reply.started":"2023-03-20T14:51:22.243887Z","shell.execute_reply":"2023-03-20T14:51:22.257611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow_io","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:49:39.042226Z","iopub.execute_input":"2023-03-20T14:49:39.042942Z","iopub.status.idle":"2023-03-20T14:49:49.511181Z","shell.execute_reply.started":"2023-03-20T14:49:39.042901Z","shell.execute_reply":"2023-03-20T14:49:49.509741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_io as tfio","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:49:49.513794Z","iopub.execute_input":"2023-03-20T14:49:49.51414Z","iopub.status.idle":"2023-03-20T14:49:49.579001Z","shell.execute_reply.started":"2023-03-20T14:49:49.514106Z","shell.execute_reply":"2023-03-20T14:49:49.577858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloader","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport librosa\nimport numpy as np\nimport tensorflow_io as tfio\n\nimport tensorflow as tf\nimport librosa\nimport numpy as np\nimport tensorflow_io as tfio\n\nclass DataLoader:        \n    def __init__(self, csv_path, train_audio_folder, sr=32000, duration=10,audio_length = 10 * 32000,window_size=1024, n_mels=128, n_fft=2048, hop_size=512, batch_size=32, shuffle=True):\n        self.train_audio_folder = train_audio_folder\n        self.csv_path = csv_path\n        self.sr = sr\n        self.duration = duration\n        self.audio_length = audio_length\n        self.n_mels = n_mels\n        self.n_fft = n_fft\n        self.hop_size = hop_size\n        self.batch_size = batch_size\n        self.window_size = window_size\n        self.shuffle = shuffle\n        self.df = pd.read_csv(self.csv_path)\n        # Create a list of image paths and labels for train, validation, and test datasets\n        self.audio_paths = [os.path.join(self.train_audio_folder, filename) for filename in self.df.filename] \n        self.classes = self.df.primary_label.values\n        self.class_to_idx = {cls_name: i for i, cls_name in enumerate(set(self.classes))}\n        self.labels = [self.class_to_idx[cls_name] for cls_name in self.classes]\n        self.labels = tf.one_hot(self.labels, 264)\n\n\n    def load_audio(self,file_path, label):\n        audio = tf.io.read_file(file_path)\n        audio = tfio.audio.decode_vorbis(audio)\n        audio = tf.cast(audio, tf.float32)\n        audio = tf.squeeze(audio, axis=-1)\n        audio = CropOrPad(audio, self.audio_length)\n        if CFG.normalize:\n            audio = Normalize(audio)\n            \n        return audio, label\n    def audio2spectrogram(self,audio, label):\n        spectrogram = tf.signal.stft(audio, frame_length=2048, frame_step=512)\n        spectrogram = tf.abs(spectrogram)\n        return spectrogram, label\n    \n    def Spectrogram2Img(self, spectrogram, label):\n        spectrogram = tf.expand_dims(spectrogram, axis=-1)\n        spectrogram = tf.image.resize(spectrogram, [224, 224])\n        spectrogram = tf.image.grayscale_to_rgb(spectrogram) \n        return spectrogram, label  \n\n    def get_dataset(self):\n        dataset = tf.data.Dataset.from_tensor_slices((self.audio_paths, self.labels))\n        dataset = dataset.map(self.load_audio, num_parallel_calls=tf.data.AUTOTUNE)\n        dataset = dataset.map(AudioAug, num_parallel_calls=tf.data.AUTOTUNE)\n        dataset = dataset.map(self.audio2spectrogram , num_parallel_calls=tf.data.AUTOTUNE)\n        dataset = dataset.map(SpecAug, num_parallel_calls=tf.data.AUTOTUNE)\n        dataset = dataset.map(self.Spectrogram2Img, num_parallel_calls=tf.data.AUTOTUNE)\n\n        if self.shuffle:\n            dataset = dataset.shuffle(buffer_size=500, reshuffle_each_iteration=True)\n        dataset = dataset.batch(batch_size=self.batch_size)\n        dataset = dataset.prefetch(tf.data.AUTOTUNE)\n        return dataset","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:55:47.480385Z","iopub.execute_input":"2023-03-20T14:55:47.481112Z","iopub.status.idle":"2023-03-20T14:55:47.499678Z","shell.execute_reply.started":"2023-03-20T14:55:47.481074Z","shell.execute_reply":"2023-03-20T14:55:47.498635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Metrices","metadata":{}},{"cell_type":"code","source":"import sklearn.metrics\n\ndef get_metrics():\n#     acc = tf.keras.metrics.BinaryAccuracy(name='acc')\n    auc = tf.keras.metrics.AUC(curve='PR', name='auc', multi_label=False) # auc on prcision-recall curve\n    acc = tf.keras.metrics.CategoricalAccuracy(name='acc')\n    return [acc, auc]\n\ndef padded_cmap(y_true, y_pred, padding_factor=5):\n    num_classes = y_true.shape[1]\n    pad_rows = np.array([[1]*num_classes]*padding_factor)\n    y_true = np.concatenate([y_true, pad_rows])\n    y_pred = np.concatenate([y_pred, pad_rows])\n    score = sklearn.metrics.average_precision_score(y_true, y_pred, average='macro',)\n    return score\n\ndef get_loss():\n    loss = tf.keras.losses.CategoricalCrossentropy(label_smoothing=CFG.label_smoothing)\n    return loss\n    \ndef get_optimizer():\n    opt = tf.keras.optimizers.Adam(learning_rate=CFG.learning_rate)\n    return opt","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:52:21.600548Z","iopub.execute_input":"2023-03-20T14:52:21.601336Z","iopub.status.idle":"2023-03-20T14:52:21.611174Z","shell.execute_reply.started":"2023-03-20T14:52:21.601293Z","shell.execute_reply":"2023-03-20T14:52:21.609749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Efficientnet","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\n# Efficient net without augmentation4\nclass EfficientNetModel:\n    \n    def __init__(self, input_shape, num_classes):\n        self.input_shape = input_shape\n        self.num_classes = num_classes\n        self.model = self.build_model()\n        \n    def build_model(self):\n        base_model = tf.keras.applications.efficientnet.EfficientNetB0(input_shape=self.input_shape, include_top=False, weights='imagenet')\n        for layer in base_model.layers[:30]:\n            layer.trainable = False\n        \n        x = base_model.output\n        \n        x = tf.keras.layers.GlobalAveragePooling2D()(x)\n        predictions = tf.keras.layers.Dense(self.num_classes, activation='softmax')(x)\n        model = tf.keras.models.Model(inputs=base_model.input, outputs=predictions)\n        return model\n     \n    def compile(self,optimizer,loss,metrics,):\n        self.model.compile(optimizer=optimizer,loss=loss, metrics=metrics)\n        \n    def train(self, train_data=None, val_data = None, epochs=10, batch_size=32, model_save_path='path'):\n        # self.model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate =learning_rate), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n        early_stop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, verbose=1, mode='min', restore_best_weights=True)\n        checkpoint = tf.keras.callbacks.ModelCheckpoint(model_save_path , monitor='val_loss', mode='min', save_weights_only=True,save_best_only=True, verbose=1)\n        ReduceLR = tf.keras.callbacks.ReduceLROnPlateau(monitpr = 'val_loss',factor=0.1,patience=2, verbose=1)\n        callbacks = [ReduceLR, early_stop, checkpoint]\n        history = self.model.fit(train_data, \n                                 epochs=epochs,\n                                 batch_size=batch_size, \n                                 validation_data=val_data, \n                                 callbacks=callbacks\n                                )\n        return history\n    def evaluate(self, data):\n        loss, accuracy = self.model.evaluate(data)\n        return loss, accuracy\n    \n    def predict(self, data):\n        return self.model.predict(data)\n    \n    def summary(self):\n        return self.model.summary()\n    \n    def save_model(self, filepath):\n        self.model.save(filepath)\n        \n    def load_model(self, model_path):\n        self.model.load_weights(model_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:02:07.087079Z","iopub.execute_input":"2023-03-20T15:02:07.087696Z","iopub.status.idle":"2023-03-20T15:02:07.100554Z","shell.execute_reply.started":"2023-03-20T15:02:07.087655Z","shell.execute_reply":"2023-03-20T15:02:07.099486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train","metadata":{}},{"cell_type":"code","source":"train_dataloader = DataLoader(csv_path = '/kaggle/working/train_df.csv', train_audio_folder = CFG.train_audio_folder)\ntrain_ds = train_dataloader.get_dataset()\n\nprint(\"Train DS\")\nfor i,(x,y) in enumerate(train_ds):\n    print(x.shape, y.shape)\n    if i==5:\n        break\n\nval_dataloader = DataLoader(csv_path = '/kaggle/working/train_df.csv', train_audio_folder = CFG.train_audio_folder)\nval_ds = val_dataloader.get_dataset()\nprint(\"Val DS\")\nfor i,(x,y) in enumerate(val_ds):\n    print(x.shape, y.shape)\n    if i==5:\n        break\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:58:34.318357Z","iopub.execute_input":"2023-03-20T14:58:34.318767Z","iopub.status.idle":"2023-03-20T15:00:22.779204Z","shell.execute_reply.started":"2023-03-20T14:58:34.318733Z","shell.execute_reply":"2023-03-20T15:00:22.778111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = EfficientNetModel(input_shape = CFG.input_shape, \n                                    num_classes = CFG.num_classes)\n\n\nmodel.compile(optimizer=get_optimizer(),\n              loss=get_loss(),\n              metrics=get_metrics())\nprint(f\"INFO======Model Compiled====\")\n# print(model.summary())\nif os.path.isfile('/kaggle/working/model.h5'):\n    print(\"INFO ===========Running the Partially Trained Model===============\")\n    model.load_model('/kaggle/working/model.h5')\n    history = model.train(train_data = train_ds,\n                          val_data =  val_ds,\n                          epochs = CFG.epochs,\n                          batch_size= CFG.batch_size,\n                          model_save_path= f\"/kaggle/working/model.h5\",\n                        )\nelse:\n    print(\"INFO ===========Running the Training of Model from Scratch===============\")\n    # model.compile(learning_rate= cfg.HyperParameter.learning_rate)\n    history = model.train(train_data = train_ds,\n                          val_data =  val_ds,\n                          epochs = CFG.epochs,\n                          batch_size= CFG.batch_size,\n                          model_save_path= f\"/kaggle/working/model.h5\",\n                        )\n    \nprint(f\"INFO ===========Training Finished===============\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T15:02:41.683358Z","iopub.execute_input":"2023-03-20T15:02:41.684071Z","iopub.status.idle":"2023-03-20T19:45:16.632786Z","shell.execute_reply.started":"2023-03-20T15:02:41.68403Z","shell.execute_reply":"2023-03-20T19:45:16.622438Z"},"trusted":true},"execution_count":null,"outputs":[]}]}