{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"  # \"jax\" or \"tensorflow\" or \"torch\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport numpy as np \nimport pandas as pd\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-16T05:19:46.256188Z","iopub.execute_input":"2024-06-16T05:19:46.256842Z","iopub.status.idle":"2024-06-16T05:20:04.409718Z","shell.execute_reply.started":"2024-06-16T05:19:46.256813Z","shell.execute_reply":"2024-06-16T05:20:04.408689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TensorFlow:\", tf.__version__)\nprint(\"Keras:\", keras.__version__)\nprint(\"KerasCV:\", keras_cv.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:20:04.411147Z","iopub.execute_input":"2024-06-16T05:20:04.411668Z","iopub.status.idle":"2024-06-16T05:20:04.416907Z","shell.execute_reply.started":"2024-06-16T05:20:04.411642Z","shell.execute_reply":"2024-06-16T05:20:04.416026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 10\n    preset = 'efficientnetv2_b2_imagenet'\n    \n    # Data augmentation parameters\n    augment=True\n\n    # Class Labels for BirdCLEF 24\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:20:08.319980Z","iopub.execute_input":"2024-06-16T05:20:08.320337Z","iopub.status.idle":"2024-06-16T05:20:08.374426Z","shell.execute_reply.started":"2024-06-16T05:20:08.320308Z","shell.execute_reply":"2024-06-16T05:20:08.373433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:20:13.739512Z","iopub.execute_input":"2024-06-16T05:20:13.740351Z","iopub.status.idle":"2024-06-16T05:20:13.744635Z","shell.execute_reply.started":"2024-06-16T05:20:13.740318Z","shell.execute_reply":"2024-06-16T05:20:13.743734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:20:54.497626Z","iopub.execute_input":"2024-06-16T05:20:54.497982Z","iopub.status.idle":"2024-06-16T05:20:54.502354Z","shell.execute_reply.started":"2024-06-16T05:20:54.497956Z","shell.execute_reply":"2024-06-16T05:20:54.501424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f'{BASE_PATH}/train_metadata.csv')\ndf['filepath'] = BASE_PATH + '/train_audio/' + df.filename\ndf['target'] = df.primary_label.map(CFG.name2label)\ndf['filename'] = df.filepath.map(lambda x: x.split('/')[-1])\ndf['xc_id'] = df.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n# Display rwos\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:20:56.441558Z","iopub.execute_input":"2024-06-16T05:20:56.442381Z","iopub.status.idle":"2024-06-16T05:20:56.708046Z","shell.execute_reply.started":"2024-06-16T05:20:56.442348Z","shell.execute_reply":"2024-06-16T05:20:56.707115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio(filepath):\n    audio, sr = librosa.load(filepath)\n    return audio, sr\n\ndef get_spectrogram(audio):\n    spec = librosa.feature.melspectrogram(y=audio, \n                                   sr=CFG.sample_rate, \n                                   n_mels=256,\n                                   n_fft=2048,\n                                   hop_length=512,\n                                   fmax=CFG.fmax,\n                                   fmin=CFG.fmin,\n                                   )\n    spec = librosa.power_to_db(spec, ref=1.0)\n    min_ = spec.min()\n    max_ = spec.max()\n    if max_ != min_:\n        spec = (spec - min_)/(max_ - min_)\n    return spec\n\ndef display_audio(row):\n    # Caption for viz\n    caption = f'Id: {row.filename} | Name: {row.common_name} | Sci.Name: {row.scientific_name} | Rating: {row.rating}'\n    # Read audio file\n    audio, sr = load_audio(row.filepath)\n    # Keep fixed length audio\n    audio = audio[:CFG.audio_len]\n    # Spectrogram from audio\n    spec = get_spectrogram(audio)\n    # Display audio\n    print(\"# Audio:\")\n    display(ipd.Audio(audio, rate=CFG.sample_rate))\n    print('# Visualization:')\n    fig, ax = plt.subplots(2, 1, figsize=(12, 2*3), sharex=True, tight_layout=True)\n    fig.suptitle(caption)\n    # Waveplot\n    lid.waveshow(audio,\n                 sr=CFG.sample_rate,\n                 ax=ax[0],\n                 color= cmap(0.1))\n    # Specplot\n    lid.specshow(spec, \n                 sr = CFG.sample_rate, \n                 hop_length=512,\n                 n_fft=2048,\n                 fmin=CFG.fmin,\n                 fmax=CFG.fmax,\n                 x_axis = 'time', \n                 y_axis = 'mel',\n                 cmap = 'coolwarm',\n                 ax=ax[1])\n    ax[0].set_xlabel('');\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:20:59.900179Z","iopub.execute_input":"2024-06-16T05:20:59.900797Z","iopub.status.idle":"2024-06-16T05:20:59.914353Z","shell.execute_reply.started":"2024-06-16T05:20:59.900763Z","shell.execute_reply":"2024-06-16T05:20:59.913523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import required packages\nfrom sklearn.model_selection import train_test_split\n\ntrain_df, valid_df = train_test_split(df, test_size=0.2)\n\nprint(f\"Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:21:03.943776Z","iopub.execute_input":"2024-06-16T05:21:03.944135Z","iopub.status.idle":"2024-06-16T05:21:03.968491Z","shell.execute_reply.started":"2024-06-16T05:21:03.944106Z","shell.execute_reply":"2024-06-16T05:21:03.967453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decodes Audio\ndef build_decoder(with_labels=True, dim=1024):\n    def get_audio(filepath):\n        file_bytes = tf.io.read_file(filepath)\n        audio = tfio.audio.decode_vorbis(file_bytes)  # decode .ogg file\n        audio = tf.cast(audio, tf.float32)\n        if tf.shape(audio)[1] > 1:  # stereo -> mono\n            audio = audio[..., 0:1]\n        audio = tf.squeeze(audio, axis=-1)\n        return audio\n\n    def crop_or_pad(audio, target_len, pad_mode=\"constant\"):\n        audio_len = tf.shape(audio)[0]\n        diff_len = abs(\n            target_len - audio_len\n        )  # find difference between target and audio length\n        if audio_len < target_len:  # do padding if audio length is shorter\n            pad1 = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            pad2 = diff_len - pad1\n            audio = tf.pad(audio, paddings=[[pad1, pad2]], mode=pad_mode)\n        elif audio_len > target_len:  # do cropping if audio length is larger\n            idx = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            audio = audio[idx : (idx + target_len)]\n        return tf.reshape(audio, [target_len])\n\n    def apply_preproc(spec):\n        # Standardize\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        # Normalize using Min-Max\n        min_val = tf.math.reduce_min(spec)\n        max_val = tf.math.reduce_max(spec)\n        spec = tf.where(\n            tf.math.equal(max_val - min_val, 0),\n            spec - min_val,\n            (spec - min_val) / (max_val - min_val),\n        )\n        return spec\n\n    def get_target(target):\n        target = tf.reshape(target, [1])\n        target = tf.cast(tf.one_hot(target, CFG.num_classes), tf.float32)\n        target = tf.reshape(target, [CFG.num_classes])\n        return target\n\n    def decode(path):\n        # Load audio file\n        audio = get_audio(path)\n        # Crop or pad audio to keep a fixed length\n        audio = crop_or_pad(audio, dim)\n        # Audio to Spectrogram\n        spec = keras.layers.MelSpectrogram(\n            num_mel_bins=CFG.img_size[0],\n            fft_length=CFG.nfft,\n            sequence_stride=CFG.hop_length,\n            sampling_rate=CFG.sample_rate,\n        )(audio)\n        # Apply normalization and standardization\n        spec = apply_preproc(spec)\n        # Spectrogram to 3 channel image (for imagenet)\n        spec = tf.tile(spec[..., None], [1, 1, 3])\n        spec = tf.reshape(spec, [*CFG.img_size, 3])\n        return spec\n\n    def decode_with_labels(path, label):\n        label = get_target(label)\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode\n","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:21:06.520754Z","iopub.execute_input":"2024-06-16T05:21:06.521279Z","iopub.status.idle":"2024-06-16T05:21:06.538470Z","shell.execute_reply.started":"2024-06-16T05:21:06.521242Z","shell.execute_reply":"2024-06-16T05:21:06.537554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_augmenter():\n    augmenters = [\n        keras_cv.layers.MixUp(alpha=0.4),\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                     width_factor=(0.06, 0.12)), # time-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                     width_factor=(1.0, 1.0)), # freq-masking\n    ]\n    \n    def augment(img, label):\n        data = {\"images\":img, \"labels\":label}\n        for augmenter in augmenters:\n            if tf.random.uniform([]) < 0.35:\n                data = augmenter(data, training=True)\n        return data[\"images\"], data[\"labels\"]\n    \n    return augment","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:21:14.068724Z","iopub.execute_input":"2024-06-16T05:21:14.069432Z","iopub.status.idle":"2024-06-16T05:21:14.076398Z","shell.execute_reply.started":"2024-06-16T05:21:14.069402Z","shell.execute_reply":"2024-06-16T05:21:14.075340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_dataset(paths, labels=None, batch_size=32, \n                  decode_fn=None, augment_fn=None, cache=True,\n                  augment=False, shuffle=2048):\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None, dim=CFG.audio_len)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter()\n        \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths,) if labels is None else (paths, labels)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache() if cache else ds\n    if shuffle:\n        opt = tf.data.Options()\n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=True)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:21:18.536479Z","iopub.execute_input":"2024-06-16T05:21:18.537225Z","iopub.status.idle":"2024-06-16T05:21:18.546757Z","shell.execute_reply.started":"2024-06-16T05:21:18.537192Z","shell.execute_reply":"2024-06-16T05:21:18.545575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train\ntrain_paths = train_df.filepath.values\ntrain_labels = train_df.target.values\ntrain_ds = build_dataset(train_paths, train_labels, batch_size=CFG.batch_size,\n                         shuffle=True, augment=CFG.augment)\n\n# Valid\nvalid_paths = valid_df.filepath.values\nvalid_labels = valid_df.target.values\nvalid_ds = build_dataset(valid_paths, valid_labels, batch_size=CFG.batch_size,\n                         shuffle=False, augment=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:21:22.282229Z","iopub.execute_input":"2024-06-16T05:21:22.282706Z","iopub.status.idle":"2024-06-16T05:21:27.065797Z","shell.execute_reply.started":"2024-06-16T05:21:22.282673Z","shell.execute_reply":"2024-06-16T05:21:27.064741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n\n\nbase_model = VGG16(weights='imagenet', include_top=False, input_shape=(None, None, 3))\n\n\nbase_model.trainable = True\n\n\nmodel = Sequential([\n    base_model,\n    GlobalAveragePooling2D(),\n    Dense(512, activation='relu'),\n    Dropout(0.3),\n    Dense(256, activation='relu'),\n    Dropout(0.3),\n    Dense(CFG.num_classes, activation='softmax')\n])\n\n\nmodel.compile(optimizer=Adam(learning_rate=0.0001), loss='categorical_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:22:20.257777Z","iopub.execute_input":"2024-06-16T05:22:20.258550Z","iopub.status.idle":"2024-06-16T05:22:20.470276Z","shell.execute_reply.started":"2024-06-16T05:22:20.258519Z","shell.execute_reply":"2024-06-16T05:22:20.469293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:22:24.392731Z","iopub.execute_input":"2024-06-16T05:22:24.393553Z","iopub.status.idle":"2024-06-16T05:22:24.398113Z","shell.execute_reply.started":"2024-06-16T05:22:24.393517Z","shell.execute_reply":"2024-06-16T05:22:24.397012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_ds, epochs=100, validation_data=valid_ds, callbacks=[early_stopping])","metadata":{"execution":{"iopub.status.busy":"2024-06-16T05:22:28.990822Z","iopub.execute_input":"2024-06-16T05:22:28.991872Z","iopub.status.idle":"2024-06-16T08:25:13.078152Z","shell.execute_reply.started":"2024-06-16T05:22:28.991840Z","shell.execute_reply":"2024-06-16T08:25:13.077071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-16T08:25:13.080052Z","iopub.execute_input":"2024-06-16T08:25:13.080373Z","iopub.status.idle":"2024-06-16T08:25:13.106208Z","shell.execute_reply.started":"2024-06-16T08:25:13.080328Z","shell.execute_reply":"2024-06-16T08:25:13.105303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-16T08:25:13.107446Z","iopub.execute_input":"2024-06-16T08:25:13.107801Z","iopub.status.idle":"2024-06-16T08:25:13.668060Z","shell.execute_reply.started":"2024-06-16T08:25:13.107770Z","shell.execute_reply":"2024-06-16T08:25:13.667122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}