{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":179935444,"sourceType":"kernelVersion"},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598},{"sourceId":56720,"sourceType":"modelInstanceVersion","modelInstanceId":47585}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport librosa\nimport librosa.display as lid\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\ncmap = mpl.colormaps['coolwarm']\nimport pandas as pd\n\nfrom tqdm.auto import tqdm\nimport os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"  # \"jax\" or \"tensorflow\" or \"torch\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-31T02:07:27.054588Z","iopub.execute_input":"2024-05-31T02:07:27.055420Z","iopub.status.idle":"2024-05-31T02:07:27.061740Z","shell.execute_reply.started":"2024-05-31T02:07:27.055383Z","shell.execute_reply":"2024-05-31T02:07:27.060746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `TrainConfig` for the CNN fine-tuning","metadata":{}},{"cell_type":"code","source":"DATA_PATH = '/kaggle/input/birdclef-2024/'\nAUDIO_PATH = '/kaggle/input/birdclef-2024/train_audio/'\n\nclass TrainConfig:\n    def __init__(self, preset):\n        self.preset = preset\n\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 10    \n    \n    # Data augmentation parameters\n    augment = True\n\n    # Class Labels for BirdCLEF 24\n    class_names = sorted(os.listdir(AUDIO_PATH))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}    ","metadata":{"execution":{"iopub.status.busy":"2024-05-31T02:01:07.884284Z","iopub.execute_input":"2024-05-31T02:01:07.885111Z","iopub.status.idle":"2024-05-31T02:01:07.915098Z","shell.execute_reply.started":"2024-05-31T02:01:07.885074Z","shell.execute_reply":"2024-05-31T02:01:07.914379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `AudioDataset`\nutility class for loading and processing audio metadata and files","metadata":{}},{"cell_type":"code","source":"class AudioDataset:\n    def __init__(self, trainconfig):\n        self.CFG = trainconfig\n\n    def load_audio(self, filepath):    \n        audio, sr = librosa.load(filepath)\n        return audio, sr\n\n    def load_metadata(self, data_path):\n        df = pd.read_csv(f'{data_path}/train_metadata.csv')\n        df['filepath'] = data_path + '/train_audio/' + df.filename\n        df['target'] = df.primary_label.map(self.CFG.name2label)\n        df['filename'] = df.filepath.map(lambda x: x.split('/')[-1])\n        df['xc_id'] = df.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n        \n        # save for future referencing\n        self.df = df\n        return df\n\n    def get_spectrogram(self, audio):\n        spec = librosa.feature.melspectrogram(\n            y=audio,     \n            sr=self.CFG.sample_rate, # sampling rate\n            n_mels=256,      # number of mel bands\n            n_fft=2048,      # length of FFT window\n            hop_length=512,  # number of samples b/w successive frames\n            fmax=self.CFG.fmax,   # max frequency\n            fmin=self.CFG.fmin    # min frequency\n        )\n        # Convert a power spectrogram (amplitude squared) to decibel (dB) units\n        spec = librosa.power_to_db(spec, ref=1.0)\n        min_ = spec.min()\n        max_ = spec.max()\n        if max_ != min_:\n            spec = (spec - min_)/(max_ - min_)\n        return spec\n\n    def display_audio(self, row):\n        # Caption for viz\n        caption = f'ID: {row.filename} | Name: {row.common_name} | Sci.Name: {row.scientific_name} | Rating: {row.rating}'\n        # Read audio file\n        audio, sr = self.load_audio(row.filepath)\n        # Keep fixed length audio\n        audio = audio[:self.CFG.audio_len]\n        # Spectrogram from audio\n        spec = self.get_spectrogram(audio)\n        # Display audio\n        # print(\"# Audio:\")\n        # playsound(row.filepath)\n        # display(ipd.Audio(audio, rate=self.CFG.sample_rate))\n        print('# Visualization:')\n        fig, ax = plt.subplots(2, 1, figsize=(12, 2*3), sharex=True, tight_layout=True)\n        fig.suptitle(caption)\n        # Waveplot\n        lid.waveshow(audio,\n                    sr=self.CFG.sample_rate,\n                    ax=ax[0],\n                    color= cmap(0.1))\n        # Specplot\n        lid.specshow(spec, \n                    sr = self.CFG.sample_rate, \n                    hop_length=512,\n                    n_fft=2048,\n                    fmin=self.CFG.fmin,\n                    fmax=self.CFG.fmax,\n                    x_axis = 'time', \n                    y_axis = 'mel',\n                    cmap = 'coolwarm',\n                    ax=ax[1])\n        ax[0].set_xlabel('');\n        fig.show()\n        plt.show()\n\n    def build_decoder(self, with_labels=True, dim=1024):\n        def get_audio(filepath):\n            # load audio and convert to float32\n            file_bytes = tf.io.read_file(filepath)\n            audio = tfio.audio.decode_vorbis(file_bytes)  # decode .ogg file\n            audio = tf.cast(audio, tf.float32)\n            if tf.shape(audio)[1] > 1:  # stereo -> mono\n                audio = audio[..., 0:1]\n            audio = tf.squeeze(audio, axis=-1)\n            return audio\n\n        def crop_or_pad(audio, target_len, pad_mode=\"constant\"):\n            audio_len = tf.shape(audio)[0]\n            diff_len = abs(\n                target_len - audio_len\n            )  # find difference between target and audio length\n            if audio_len < target_len:  # do padding if audio length is shorter\n                pad1 = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n                pad2 = diff_len - pad1\n                audio = tf.pad(audio, paddings=[[pad1, pad2]], mode=pad_mode)\n            elif audio_len > target_len:  # do cropping if audio length is larger\n                idx = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n                audio = audio[idx : (idx + target_len)]\n            return tf.reshape(audio, [target_len])\n\n        def apply_preproc(spec):\n            # Standardize\n            mean = tf.math.reduce_mean(spec)\n            std = tf.math.reduce_std(spec)\n            spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n            # Normalize using Min-Max\n            min_val = tf.math.reduce_min(spec)\n            max_val = tf.math.reduce_max(spec)\n            spec = tf.where(\n                tf.math.equal(max_val - min_val, 0),\n                spec - min_val,\n                (spec - min_val) / (max_val - min_val),\n            )\n            return spec\n\n        def get_target(target):\n            target = tf.reshape(target, [1])\n            target = tf.cast(tf.one_hot(target, self.CFG.num_classes), tf.float32)\n            target = tf.reshape(target, [self.CFG.num_classes])\n            return target\n\n        def decode(path):\n            # Load audio file\n            audio = get_audio(path)\n            # Crop or pad audio to keep a fixed length\n            audio = crop_or_pad(audio, dim)\n            # Audio to Spectrogram\n            spec = keras.layers.MelSpectrogram(\n                num_mel_bins=self.CFG.img_size[0],\n                fft_length=self.CFG.nfft,\n                sequence_stride=self.CFG.hop_length,\n                sampling_rate=self.CFG.sample_rate,\n            )(audio)\n            # Apply normalization and standardization\n            spec = apply_preproc(spec)\n            # Spectrogram to 3 channel image (for imagenet)\n            spec = tf.tile(spec[..., None], [1, 1, 3])\n            spec = tf.reshape(spec, [*self.CFG.img_size, 3])\n            return spec\n\n        def decode_with_labels(path, label):\n            label = get_target(label)\n            return decode(path), label\n\n        return decode_with_labels if with_labels else decode\n\n    def build_augmenter(self):\n        # these seem quite clever and domain-specific\n        augmenters = [\n            keras_cv.layers.MixUp(alpha=0.4),\n            keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                        width_factor=(0.06, 0.12)), # time-masking (x-axis)\n            keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                        width_factor=(1.0, 1.0)), # freq-masking (y-axis)\n        ]\n        \n        def augment(img, label):\n            data = {\"images\":img, \"labels\":label}\n            for augmenter in augmenters:\n                if tf.random.uniform([]) < 0.35:\n                    data = augmenter(data, training=True)\n            return data[\"images\"], data[\"labels\"]\n        \n        return augment\n\n    def build_dataset(self, paths, labels=None, batch_size=32, \n                    decode_fn=None, augment_fn=None, cache=True,\n                    augment=False, shuffle=2048):\n\n        if decode_fn is None:\n            decode_fn = self.build_decoder(labels is not None, dim=self.CFG.audio_len)\n\n        if augment_fn is None:\n            augment_fn = self.build_augmenter()\n            \n        AUTO = tf.data.experimental.AUTOTUNE\n        slices = (paths,) if labels is None else (paths, labels)\n        ds = tf.data.Dataset.from_tensor_slices(slices)\n        ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n        ds = ds.cache() if cache else ds\n        if shuffle:\n            opt = tf.data.Options()\n            ds = ds.shuffle(shuffle, seed=self.CFG.seed)\n            opt.experimental_deterministic = False\n            ds = ds.with_options(opt)\n        ds = ds.batch(batch_size, drop_remainder=True)\n        ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n        ds = ds.prefetch(AUTO)\n        return ds","metadata":{"execution":{"iopub.status.busy":"2024-05-31T02:01:09.754461Z","iopub.execute_input":"2024-05-31T02:01:09.754811Z","iopub.status.idle":"2024-05-31T02:01:09.789444Z","shell.execute_reply.started":"2024-05-31T02:01:09.754780Z","shell.execute_reply":"2024-05-31T02:01:09.788607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 🤖 Model setup","metadata":{}},{"cell_type":"code","source":"# load fine-tuned model\npreset = \"efficientnetv2_b2_imagenet\"\nconfig_efficientnet = TrainConfig(preset)\n\ninp = keras.layers.Input(shape=(None, None, 3))\nbackbone = keras_cv.models.EfficientNetV2Backbone.from_preset(\n    config_efficientnet.preset\n)\nout = keras_cv.models.ImageClassifier(\n    backbone=backbone,\n    num_classes=TrainConfig.num_classes,\n    name=\"classifier\"\n)(inp)\nmodel = keras.models.Model(inputs=inp, outputs=out)\n\nft_weights_path = \"/kaggle/input/efficientnet-v2-birdclef-2024-finetune/keras/v1/1/best_model.weights.h5\"\nmodel.load_weights(ft_weights_path)\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2024-05-31T02:01:44.441788Z","iopub.execute_input":"2024-05-31T02:01:44.442494Z","iopub.status.idle":"2024-05-31T02:01:51.510632Z","shell.execute_reply.started":"2024-05-31T02:01:44.442459Z","shell.execute_reply":"2024-05-31T02:01:51.509742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modified from:\n# https://github.com/jonaac/deep-xgboost-image-classifier/blob/main/code/resnet/cnn_resnet_xgboost.py#L51\ndef extract_feature_layer(model):\n    # constructed model is an input layer tacked onto a ImageClassifier, need to grab classifier\n    clf = model.layers[1]\n    total_layers = len(clf.layers)\n    fl_index = -2\n\n    # construct a new model\n    return keras.Model(\n        inputs=clf.input,\n        outputs=clf.get_layer(index=fl_index).output\n    )\n\ndef cnn_vectorize(feature_layer_model, data):    \n    feature_layer_output = feature_layer_model.predict(data, verbose=False)\n    return feature_layer_output","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:42:12.374512Z","iopub.execute_input":"2024-05-30T22:42:12.374790Z","iopub.status.idle":"2024-05-30T22:42:12.380893Z","shell.execute_reply.started":"2024-05-30T22:42:12.374767Z","shell.execute_reply":"2024-05-30T22:42:12.379715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"birdclef = AudioDataset(config_efficientnet)\ndf = birdclef.load_metadata(DATA_PATH)\n\n# print(df.head(2))\ntrain_df, valid_df = train_test_split(df, test_size=0.2)\nprint(f\"Total: {len(df):,} | Num Train: {len(train_df):,} | Num Valid: {len(valid_df):,}\")\n\n# build train and test dataloaders\ntrain_paths = train_df.filepath.values\ntrain_labels = train_df.target.values\ntrain_ds = birdclef.build_dataset(\n    train_paths,\n    train_labels,\n    batch_size=config_efficientnet.batch_size,\n    shuffle=True, \n    augment=TrainConfig.augment\n)\nvalid_paths = valid_df.filepath.values\nvalid_labels = valid_df.target.values\nvalid_ds = birdclef.build_dataset(\n    valid_paths,\n    valid_labels,\n    batch_size=config_efficientnet.batch_size,\n    shuffle=False,\n    augment=False\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-31T02:04:09.800514Z","iopub.execute_input":"2024-05-31T02:04:09.800964Z","iopub.status.idle":"2024-05-31T02:04:14.007656Z","shell.execute_reply.started":"2024-05-31T02:04:09.800929Z","shell.execute_reply":"2024-05-31T02:04:14.006600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_size = train_df.shape[0]\nval_size = valid_df.shape[0]\n# extract last layer of CNN -- will be called repeatedly\nfeature_layer_model = extract_feature_layer(model)\nvectorizer_output_size = feature_layer_model.layers[-1].output.shape[-1]\n\n# collect X_train and y_train for XGBoost here\nX_train = np.zeros(shape=(train_size, vectorizer_output_size))\ny_train = np.zeros(shape=(train_size, TrainConfig.num_classes))\nprint('Vectorizing train data for XGBoost')\nfor i, (batch, labels) in enumerate(tqdm(train_ds)):\n    curr_batch_size = batch.shape[0]\n    vectorized_audio = cnn_vectorize(feature_layer_model, batch)\n    \n    start_ind = i * TrainConfig.batch_size\n    X_train[start_ind : start_ind+curr_batch_size] = vectorized_audio\n    y_train[start_ind : start_ind+curr_batch_size] = labels\n    \n# same for validation data\nX_val = np.zeros(shape=(val_size, vectorizer_output_size))\ny_val = np.zeros(shape=(val_size, TrainConfig.num_classes))\nprint('Vectorizing validation data for XGBoost')\nfor i, (batch, labels) in enumerate(tqdm(valid_ds)):\n    curr_batch_size = batch.shape[0]\n    vectorized_audio = cnn_vectorize(feature_layer_model, batch)\n    \n    start_ind = i * TrainConfig.batch_size\n    X_val[start_ind : start_ind+curr_batch_size] = vectorized_audio\n    y_val[start_ind : start_ind+curr_batch_size] = labels","metadata":{"execution":{"iopub.status.busy":"2024-05-30T22:46:08.349488Z","iopub.execute_input":"2024-05-30T22:46:08.349879Z","iopub.status.idle":"2024-05-30T23:40:59.148265Z","shell.execute_reply.started":"2024-05-30T22:46:08.349847Z","shell.execute_reply":"2024-05-30T23:40:59.147262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.savetxt('X_train_xgb.txt', X_train)\nnp.savetxt('y_train_xgb.txt', y_train)\n\nnp.savetxt('X_val_xgb.txt', X_val)\nnp.savetxt('y_val_xgb.txt', y_val)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:41:57.009520Z","iopub.execute_input":"2024-05-30T23:41:57.010280Z","iopub.status.idle":"2024-05-30T23:42:30.524493Z","shell.execute_reply.started":"2024-05-30T23:41:57.010246Z","shell.execute_reply":"2024-05-30T23:42:30.523395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# last 47 are empty, don't know why -- drop them\nprint(f\"Before --- X_train: {X_train.shape}; y_train: {y_train.shape}\")\nvalid_inds_train = np.any(X_train > 0, axis=1)\nX_train_cleaned = X_train[valid_inds_train]\ny_train_cleaned = y_train[valid_inds_train]\nprint(f\"After --- X_train: {X_train_cleaned.shape}; y_train: {y_train_cleaned.shape}\")\n\nprint(f\"Before --- X_val: {X_val.shape}; y_val: {y_val.shape}\")\nvalid_inds_val = np.any(X_val > 0, axis=1)\nX_val_cleaned = X_val[valid_inds_val]\ny_val_cleaned = y_val[valid_inds_val]\nprint(f\"After --- X_val: {X_val_cleaned.shape}; y_val: {y_val_cleaned.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-30T23:51:10.442828Z","iopub.execute_input":"2024-05-30T23:51:10.443235Z","iopub.status.idle":"2024-05-30T23:51:10.573346Z","shell.execute_reply.started":"2024-05-30T23:51:10.443207Z","shell.execute_reply":"2024-05-30T23:51:10.572322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### TODO: visualize X_train with T-SNE, colored by class label","metadata":{}},{"cell_type":"markdown","source":"### Fit XGBoost","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nimport pickle","metadata":{"execution":{"iopub.status.busy":"2024-05-31T01:52:39.987571Z","iopub.execute_input":"2024-05-31T01:52:39.988177Z","iopub.status.idle":"2024-05-31T01:52:42.130222Z","shell.execute_reply.started":"2024-05-31T01:52:39.988144Z","shell.execute_reply":"2024-05-31T01:52:42.129278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dtrain = xgb.DMatrix(train_cnn_feature_layer, label=train_labels)\n# dtrain = xgb.DMatrix(X_train_cleaned, label=train_labels[valid_inds_train])\n# dtest = xgb.DMatrix(X_val_cleaned, label=valid_labels[valid_inds_val])\n# dtrain.save_binary('train.buffer')\n# dtest.save_binary('test.buffer')\n\nparams = {\n    'max_depth': 12,\n    'eta': 0.05,\n    'device': 'cuda',\n    'tree_method': 'hist',\n    'objective': 'multi:softprob',\n    'num_class': TrainConfig.num_classes,\n    'early_stopping_rounds': 5,\n    'eval_metric': 'merror'\n}\n\nwatchlist = [(dtrain, 'train'),(dtest, 'eval')]\nn_round = 175\nxgb_model = xgb.train(\n    params,\n    dtrain,\n    n_round,\n    watchlist\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-31T01:18:21.367858Z","iopub.execute_input":"2024-05-31T01:18:21.368164Z","iopub.status.idle":"2024-05-31T01:25:41.738079Z","shell.execute_reply.started":"2024-05-31T01:18:21.368138Z","shell.execute_reply":"2024-05-31T01:25:41.737237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model.save_model('efficientnet_xgb_trained.json')\n\n# xgb_model.evalpickle.dump(xgb_model, open(\"efficientnet_xgboost_final.pickle.dat\", \"wb\"))","metadata":{"execution":{"iopub.status.busy":"2024-05-31T01:29:04.644994Z","iopub.execute_input":"2024-05-31T01:29:04.645378Z","iopub.status.idle":"2024-05-31T01:29:06.444055Z","shell.execute_reply.started":"2024-05-31T01:29:04.645349Z","shell.execute_reply":"2024-05-31T01:29:06.443226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate","metadata":{}},{"cell_type":"code","source":"# if model has been trained, load these\ndtrain = xgb.DMatrix(\"train.buffer\")\ndtest = xgb.DMatrix(\"test.buffer\")\n\nX_val = np.loadtxt(\"/kaggle/working/X_val_xgb.txt\")\ny_val = np.loadtxt(\"/kaggle/working/y_val_xgb.txt\")\nprint(f\"Before --- X_val: {X_val.shape}; y_val: {y_val.shape}\")\nvalid_inds_val = np.any(X_val > 0, axis=1)\nX_val_cleaned = X_val[valid_inds_val]\ny_val_cleaned = y_val[valid_inds_val]\nprint(f\"After --- X_val: {X_val_cleaned.shape}; y_val: {y_val_cleaned.shape}\")\n\nxgb_model = xgb.Booster(model_file=\"efficientnet_xgb_trained.json\")","metadata":{"execution":{"iopub.status.busy":"2024-05-31T02:09:43.885783Z","iopub.execute_input":"2024-05-31T02:09:43.886160Z","iopub.status.idle":"2024-05-31T02:09:51.442154Z","shell.execute_reply.started":"2024-05-31T02:09:43.886128Z","shell.execute_reply":"2024-05-31T02:09:51.441377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probas = xgb_model.predict(dtest)\npredictions = np.argmax(probas, axis=1)\nprint(accuracy_score(valid_labels[valid_inds_val], predictions) * 100)","metadata":{"execution":{"iopub.status.busy":"2024-05-31T02:10:24.662105Z","iopub.execute_input":"2024-05-31T02:10:24.662781Z","iopub.status.idle":"2024-05-31T02:10:24.678123Z","shell.execute_reply.started":"2024-05-31T02:10:24.662749Z","shell.execute_reply":"2024-05-31T02:10:24.677324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}