{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8620538,"sourceType":"datasetVersion","datasetId":5160203},{"sourceId":182025443,"sourceType":"kernelVersion"},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2024 [Train]\n\n<img src = 'https://www.imageclef.org/system/files/new-banner-bird-clef.png' \n     align = 'center'/>\n\nThis notebook project is to identified bird species with bird calls, for monitor bird populations for conservation purposes.","metadata":{}},{"cell_type":"markdown","source":"### This is my first joining Kaggle competition and working with audio dataset.Thanks to Keras team showing me in right direction, I would still have struggle shaping the Audio data from Audio to shape and array's.Learn a lot about Tenserflow and Keras and how incredible this open source platforms is.I struggle to submit the notebook or there is a trick I miss to submit notebooks.Thanks enjoy this project and learn a lot about audio dataset.    \n","metadata":{}},{"cell_type":"markdown","source":"## Project Libraries","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\n\nos.environ[\"KERAS_BACKEND\"] = \"jax\"  \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nfrom glob import glob\nfrom tqdm.auto import tqdm\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\n%config InlineBackend.show_traceback = True\n\nXLA_COMPILE_TRAIN_STEP = False","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:17.621031Z","iopub.execute_input":"2024-06-07T09:50:17.621393Z","iopub.status.idle":"2024-06-07T09:50:35.996271Z","shell.execute_reply.started":"2024-06-07T09:50:17.621365Z","shell.execute_reply":"2024-06-07T09:50:35.995258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Choosing ColorMap","metadata":{}},{"cell_type":"code","source":"cmap = mpl.cm.get_cmap('inferno')","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:40.920090Z","iopub.execute_input":"2024-06-07T09:50:40.921021Z","iopub.status.idle":"2024-06-07T09:50:40.926294Z","shell.execute_reply.started":"2024-06-07T09:50:40.920990Z","shell.execute_reply":"2024-06-07T09:50:40.925166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inspecting the data of train_metadata.csv","metadata":{}},{"cell_type":"code","source":"train_meta = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\ntrain_meta.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:43.711453Z","iopub.execute_input":"2024-06-07T09:50:43.712237Z","iopub.status.idle":"2024-06-07T09:50:43.919032Z","shell.execute_reply.started":"2024-06-07T09:50:43.712195Z","shell.execute_reply":"2024-06-07T09:50:43.917919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inspecting the data of eBird_Taxonomy_v2021.csv","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv')\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:47.340369Z","iopub.execute_input":"2024-06-07T09:50:47.340775Z","iopub.status.idle":"2024-06-07T09:50:47.433913Z","shell.execute_reply.started":"2024-06-07T09:50:47.340743Z","shell.execute_reply":"2024-06-07T09:50:47.432785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration ","metadata":{}},{"cell_type":"code","source":"class config:\n    # seed for reproducibility\n    seed = 42\n    \n    # InputData\n    img_size = [128, 384]\n    duration = 5 \n    sample_rate = 32000\n    audio_length = duration * sample_rate\n    \n    # Short-Time Fourier Transform (STFT)\n    nfft = 2028\n    window = 2048\n    hop_length = audio_length // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Training\n    batch_size = 64\n    epochs = 10\n    preset = 'efficientnetv2_b2_imagenet'\n    \n    ## Augmentation\n    augment=True\n\n    # Classification parameters for class labels\n    c_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    number_c = len(c_names)\n    c_labels = list(range(number_c))\n    label_name = dict(zip(c_labels, c_names))\n    name_label = {v:k for k,v in label_name.items()}\n        ","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:50.575853Z","iopub.execute_input":"2024-06-07T09:50:50.576242Z","iopub.status.idle":"2024-06-07T09:50:50.602158Z","shell.execute_reply.started":"2024-06-07T09:50:50.576211Z","shell.execute_reply":"2024-06-07T09:50:50.601052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Seed","metadata":{}},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(config.seed)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:54.800065Z","iopub.execute_input":"2024-06-07T09:50:54.800823Z","iopub.status.idle":"2024-06-07T09:50:54.805380Z","shell.execute_reply.started":"2024-06-07T09:50:54.800788Z","shell.execute_reply":"2024-06-07T09:50:54.804362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add Columns to Meta Data","metadata":{}},{"cell_type":"code","source":"train_meta['filepath'] = '/kaggle/input/birdclef-2024/train_audio/' + train_meta.filename\ntrain_meta['filename'] = train_meta.filepath.apply(lambda x: x.split('/')[-1])\ntrain_meta['target'] = train_meta.primary_label.map(config.name_label)\ntrain_meta['xc_id'] = train_meta.filepath.apply(lambda x: x.split('/')[-1].split('.')[0])\n\ntrain_meta.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:50:58.161565Z","iopub.execute_input":"2024-06-07T09:50:58.162506Z","iopub.status.idle":"2024-06-07T09:50:58.244163Z","shell.execute_reply.started":"2024-06-07T09:50:58.162457Z","shell.execute_reply":"2024-06-07T09:50:58.243073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Audio Load and Visualize","metadata":{}},{"cell_type":"code","source":"# Load Audio \ndef load_audio(filepath):\n    audio, sr = librosa.load(filepath)\n    return audio, sr\n\n# Compute mel-scaled spectrogram\ndef get_spectrogram(audio):\n    spec = librosa.feature.melspectrogram(y = audio,\n    sr = config.sample_rate,\n    n_mels = 256,\n    n_fft = 2048,\n    hop_length = 512,\n    fmax = config.fmax,\n    fmin = config.fmin,\n    )\n    \n    # Convert to decibels\n    spec = librosa.power_to_db(spec, ref=1.0)\n    # Normalize Spectrogram  \n    min_ = spec.min()\n    max_ = spec.max()\n    if max_ != min_:\n        spec = (spec - min_) / (max_ - min_)\n    return spec\n\ndef display_audio(row):\n\n# Load audio files\n    audio, sr = load_audio(row.filepath)\n\n# Keep fixed length audio from Configuration InputData\n    audio = audio[:config.audio_length]\n\n# Spectrogram from audio\n    spec = get_spectrogram(audio)\n\n# Display audio\n    print('# Audio:')\n    display(ipd.Audio(audio, rate = config.sample_rate))\n    print('# Visualization Plots:')\n    fig = plt.subplots(figsize = (12, 3), tight_layout = True, sharex = True)\n    plt.suptitle(f'XC_ID: {row.xc_id} - Name: {row.common_name} - Science Name: {row.scientific_name} - Rating: {row.rating}')\n  \n    # Specplot\n    lid.specshow(spec,\n                 sr = config.sample_rate,\n                 hop_length = 512,\n                 n_fft = 2048,\n                 fmin = config.fmin,\n                 fmax = config.fmax,\n                 x_axis = 'time',\n                 y_axis = 'mel',\n                 cmap = 'inferno',\n                )\n\n    # Waveplot\n    fig = plt.subplots(figsize = (12, 3), tight_layout = True, sharex = True)\n    plt.suptitle(f'WavePlot - XC_ID: {row.xc_id}')\n    plt.ylabel('Amplitude')\n    \n    lid.waveshow(audio,\n                 sr = config.sample_rate,\n                 color = 'blue',\n                )            \n    \n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:01.545962Z","iopub.execute_input":"2024-06-07T09:51:01.546868Z","iopub.status.idle":"2024-06-07T09:51:01.560151Z","shell.execute_reply.started":"2024-06-07T09:51:01.546827Z","shell.execute_reply":"2024-06-07T09:51:01.558855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bird Call ID XC175797","metadata":{}},{"cell_type":"code","source":"row = train_meta[train_meta['xc_id'] == 'XC175797'].iloc[0]\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:05.043645Z","iopub.execute_input":"2024-06-07T09:51:05.044699Z","iopub.status.idle":"2024-06-07T09:51:17.995027Z","shell.execute_reply.started":"2024-06-07T09:51:05.044648Z","shell.execute_reply":"2024-06-07T09:51:17.994011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bird Call ID XC842684","metadata":{}},{"cell_type":"code","source":"row = train_meta[train_meta['xc_id'] == 'XC842684'].iloc[0]\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:21.808971Z","iopub.execute_input":"2024-06-07T09:51:21.810401Z","iopub.status.idle":"2024-06-07T09:51:23.145372Z","shell.execute_reply.started":"2024-06-07T09:51:21.810356Z","shell.execute_reply":"2024-06-07T09:51:23.143442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bird Call ID XC551992","metadata":{}},{"cell_type":"code","source":"row = train_meta[train_meta['xc_id'] == 'XC551992'].iloc[0]\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:25.740688Z","iopub.execute_input":"2024-06-07T09:51:25.741059Z","iopub.status.idle":"2024-06-07T09:51:27.002293Z","shell.execute_reply.started":"2024-06-07T09:51:25.741028Z","shell.execute_reply":"2024-06-07T09:51:27.001260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### In train_auto data .ogg files some not a lot of the .ogg files audio duration is lower as 3, 4 seconds I have choice 5 seconds for all the .ogg files to visualize the data in spectrogram. ","metadata":{}},{"cell_type":"markdown","source":"# Split the data in to train and test dataset","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nmeta_train, valid_meta = train_test_split(train_meta, test_size = 0.2)\n\nprint('Train:', len(meta_train), 'Valid:', len(valid_meta))","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:30.165439Z","iopub.execute_input":"2024-06-07T09:51:30.165826Z","iopub.status.idle":"2024-06-07T09:51:30.309780Z","shell.execute_reply.started":"2024-06-07T09:51:30.165797Z","shell.execute_reply":"2024-06-07T09:51:30.308622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decode Audio","metadata":{}},{"cell_type":"code","source":"# Decodes Audio\ndef b_decoder(with_labels = True, dim = 1024):\n    \n#     Builds decoder function for audio files.\n    \n#     Args:\n#     with labels (bool): Whether to include labels in the output.\n#     dim (int): The target length for padding and croping.\n    \n#     return:\n#     A decoder fuction that takes a file path (and optional label) as input.\n\n    def get_audio(filepath):\n        # Use load audio file\n        audio_bytes = tf.io.read_file(filepath)\n        audio = tfio.audio.decode_vorbis(audio_bytes)  \n        audio = tf.cast(audio, tf.float32)\n        # Convert audio to mono if necessary\n        if tf.shape(audio)[1] > 1:  \n            audio = audio[..., 0:1]\n        # Remove last dimensions\n        audio = tf.squeeze(audio, axis = -1)\n        return audio\n\n    def preprocess_spectrogram(spec):\n        # Standardize spectrogram value\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        # Normalize spectrogram values to [0, 1]\n        min_val = tf.math.reduce_min(spec)\n        max_val = tf.math.reduce_max(spec)\n        spec = tf.where(\n            tf.math.equal(max_val - min_val, 0),\n            spec - min_val,\n            (spec - min_val) / (max_val - min_val),\n        )\n        return spec\n    \n    def crop_pad(audio, target_length):\n        # Calculate padding and cropping amount\n        audio_length = tf.shape(audio)[0]\n        diff_length = abs(\n            target_length - audio_length\n        )  # find difference between target and audio length\n        if audio_length < target_length:\n            # Padding audio with random start and end padding\n            padding1 = tf.random.uniform([], maxval = diff_length, dtype = tf.int32)\n            padding2 = diff_length - padding1\n            audio = tf.pad(audio, paddings = [[padding1, padding2]], mode = \"constant\")\n        elif audio_length > target_length:\n            # crop audio with random start index\n            idx = tf.random.uniform([], maxval = diff_length, dtype = tf.int32)\n            audio = audio[idx : (idx + target_length)]\n        return tf.reshape(audio, [target_length])\n\n    def decode(path):\n        # Load audio file\n        audio = get_audio(path)\n        # Pad and Crop to target length\n        audio = crop_pad(audio, dim)\n        # Convert audio to spectrogram\n        spectrogram = keras.layers.MelSpectrogram(\n            num_mel_bins = config.img_size[0],\n            fft_length = config.nfft,\n            sequence_stride = config.hop_length,\n            sampling_rate = config.sample_rate,\n        )(audio)\n        # preprocess spectrogram\n        spectrogram = preprocess_spectrogram(spectrogram)\n        # Convert spectrogram to 3-channel image\n        spectrogram = tf.tile(spectrogram[..., None], [1, 1, 3])\n        spectrogram = tf.reshape(spectrogram, [*config.img_size, 3])\n        return spectrogram\n    \n    def get_target(target):\n        # One-hot encode label\n        target = tf.reshape(target, [1])\n        target = tf.cast(tf.one_hot(target, config.number_c), tf.float32)\n        target = tf.reshape(target, [config.number_c])\n        return target\n\n    def decode_with_labels(path, label):\n        # Get label\n        label = get_target(label)\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode\n            ","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:33.101840Z","iopub.execute_input":"2024-06-07T09:51:33.102261Z","iopub.status.idle":"2024-06-07T09:51:33.122031Z","shell.execute_reply.started":"2024-06-07T09:51:33.102230Z","shell.execute_reply":"2024-06-07T09:51:33.121006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Augmenter Class\n\n# Build TensorFlow Input Pipelines ","metadata":{}},{"cell_type":"code","source":"def b_augmenter():\n# Builds an augmenter function for audio data.\n    \n# Returns:\n# An augmenter fuction that takes an image and label as input.\n    \n    augmenters = [\n        # Mixup augmentation with alpha = 0.4\n        keras_cv.layers.MixUp(alpha=0.4),\n        \n        # Time masking with random cutout\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0), # no change in height\n                                     width_factor=(0.06, 0.12)), # 6-12% cutout in width (time axis)\n        \n        # Frequency masking with random cutout\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1), # 6-10% cutout in height (Frequency axis)\n                                     width_factor=(1.0, 1.0)), # no change in width\n    ]\n    \n    def augment(img, label):\n    # Applies augmentation to the input image and label.\n        \n    # Args:\n    # mg: The input image.\n    # label: The input label.\n        \n    # Returns:\n    # The augmented image and label.\n    \n        data = {\"images\":img, \"labels\":label}\n        for augmenter in augmenters:\n            if tf.random.uniform([]) < 0.35:\n            # Apply augmentation with 35% probability\n                data = augmenter(data, training=True)\n        return data[\"images\"], data[\"labels\"]\n    \n    return augment","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:39.350289Z","iopub.execute_input":"2024-06-07T09:51:39.351054Z","iopub.status.idle":"2024-06-07T09:51:39.358922Z","shell.execute_reply.started":"2024-06-07T09:51:39.351023Z","shell.execute_reply":"2024-06-07T09:51:39.357907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train and Validation loaders","metadata":{}},{"cell_type":"code","source":"def b_dataset(paths, labels = None, batch_size = 32, decode_fn = None, augment_fn = None, cache = True,\n                  augment = False, shuffle = 2048):\n#      Builds a dataset from audio files.\n    \n#      Args:\n#      Paths (list): List of file paths.\n#      labels (list, optional): List of labels, defaults to None.\n#      batch_size (int, optional): Batch size, defaults to 32.\n#      decode_ (list, optional): decoding function, defaults to b_m_decoder.\n#      augment_ (list, optional): Augmentation function, defaults to create_augmenter.\n#      cache (bool, optional): Whether to cache the dataset, defaults to True.\n#      augment (bool, optional): Whether to apply augmentation, defaults to None.\n#      shuffle (int, optional): Shuffle buffer size, defaults to 2048.\n    \n#      Return:\n#      tf.data.Dataset: The build dataset.\n\n    if decode_fn is None:\n        decode_fn = b_decoder(labels is not None, dim = config.audio_length)\n\n    if augment_fn is None:\n        augment_fn = b_augmenter()\n        \n    auto_tune = tf.data.experimental.AUTOTUNE\n    \n    ds = tf.data.Dataset.from_tensor_slices((paths,) if labels is None else (paths, labels))\n    \n    # Use `map` with `num_parallel_calls` for parallel processing\n    ds = ds.map(decode_fn, num_parallel_calls = auto_tune)\n    \n    # If cache the dataset if enabled\n    ds = ds.cache() if cache else ds\n    \n    # If shuffle the dataset if enabled.\n    if shuffle:\n        opt = tf.data.Options()\n        ds = ds.shuffle(shuffle, seed = config.seed)\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    \n    # Batch the dataset\n    ds = ds.batch(batch_size, drop_remainder = True)\n    \n    # Apply augmentation if enabled\n    ds = ds.map(augment_fn, num_parallel_calls = auto_tune) if augment else ds\n    \n    # Prefetch the dataset\n    ds = ds.prefetch(auto_tune)\n    \n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:44.127529Z","iopub.execute_input":"2024-06-07T09:51:44.127959Z","iopub.status.idle":"2024-06-07T09:51:44.139331Z","shell.execute_reply.started":"2024-06-07T09:51:44.127923Z","shell.execute_reply":"2024-06-07T09:51:44.138129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train the Dataset and Validation the Dataset","metadata":{}},{"cell_type":"code","source":"# Train dataset\ntrain_paths = meta_train.filepath.values\ntrain_labels = meta_train.target.values\ntrain_ds = b_dataset(train_paths, train_labels, batch_size = config.batch_size, shuffle = True, augment = config.augment)\n\n# Validation dataset\nvalid_paths = valid_meta.filepath.values\nvalid_labels = valid_meta.target.values\nvalid_ds = b_dataset(valid_paths, valid_labels, batch_size = config.batch_size, shuffle = False, augment = False)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:47.864758Z","iopub.execute_input":"2024-06-07T09:51:47.865389Z","iopub.status.idle":"2024-06-07T09:51:52.281524Z","shell.execute_reply.started":"2024-06-07T09:51:47.865356Z","shell.execute_reply":"2024-06-07T09:51:52.280671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Summary","metadata":{}},{"cell_type":"code","source":"# Input meaning can handle input data with any height, width, and 3 color channels (RGB)\ninput_data = keras.layers.Input(shape = (None, None, 3))\n\n# This is the EfficientNetV2B2 backbone model, which takes the input data and processes\n#  it through several convolutional and pooling layers. The output shape is (None, None, None, 1408),\n#  indicating that it produces 1408 feature maps.\nres_backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(config.preset)\n\n#Backbone\nx = res_backbone(input_data)\n\n# This is a convolutional layers of 32 filters, taking the output from the backbone model and producing \n# 32 feature maps. \nx = keras.layers.Conv2D(32, 3, activation = 'relu')(x)\n\n# This layer performs global average pooling on the output from the convolational layer, reducing the spatial\n# dimensions to produce 1D output with 32 elements \nx = keras.layers.GlobalAveragePooling2D()(x)\n\n# This is a dense (fully connected) layers with 182 units, taking the output from the global average pooling \n# layer and producing a final output of 182 elements.\nout_data = keras.layers.Dense(config.number_c, activation = 'softmax')(x)\n\n# The model takes a input layer with shape of Input function, processes it through two hidden layers, and\n# produces output layer with shape of 182. This represent a neurel network that takes a Input\n# function dimensional input and produces a 182 dimensional output.\nmodel = keras.models.Model(inputs = input_data, outputs = out_data)\n\n# Configures the model for training.\n# Optimizer 'adam': This specifies the optimization algorithm to use for training\nmodel.compile(optimizer = \"adam\",\n              # loss = keras.losses.CategoricalCrossentropy: This specifies the loss function to use for training. \n              # In this case, it's the categorical cross-entropy loss function, \n              # which is suitable for multi-class classification problems. \n              # The label_smoothing parameter is set to 0.02, which adds a small amount of noise to the labels to help prevent overfitting.\n              loss = keras.losses.CategoricalCrossentropy(label_smoothing = 0.02),\n              # metrics = [keras.metrics.AUC(name='auc')]: This specifies the metrics to track during training. \n              # In this case, it's the area under the ROC curve (AUC) metric, \n              # which is a common metric for evaluating the performance of binary classification models. \n              # The name='auc' parameter gives the metric a name, which can be useful for plotting and tracking the metric during training.\n              metrics = [keras.metrics.AUC(name = 'auc')],\n             )\n\n# Call model.summary() to print a useful summary of the model, which includes:\n# Name and type of all layers in the model.\n# Output shape for each layer.\n# Number of weight parameters of each layer.\n# If the model has general topology (discussed below), the inputs each layer receives\n# The total number of trainable and non-trainable parameters of the model.\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:51:55.214103Z","iopub.execute_input":"2024-06-07T09:51:55.215171Z","iopub.status.idle":"2024-06-07T09:52:19.240002Z","shell.execute_reply.started":"2024-06-07T09:51:55.215109Z","shell.execute_reply":"2024-06-07T09:52:19.238918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LR Schedule Graph","metadata":{}},{"cell_type":"code","source":"# Import math librarie\nimport math\n\n# plot_lr_schudule calculates and returns a learning rate schedule base on the input parameters.\n\n# Input Parameters:\n# batch_size: Batch size use for training.\n# mode: The learning rate schedule mode. Options are 'exp', 'step' and 'cos'.\n# epochs: The total numbers of training epochs.\ndef plot_lr_schudule(batch_size = 8, mode = 'cos', epochs = 10):\n    \n    # Learning rate calculations:\n    # lr_start, lr_max, lr_min is the starting,\n    # lr_start = 5e-5 is the starting learning rate, which is relatively small value. This is because\n    # the model is initaily learning the most important features and needs a smaller learning rate to\n    # avoid overshooting.\n    # lr_max = 8e-6 * batch_size: This is because the learning rate should be proportional to the batch\n    # size to ensure stable training. A larger batch size require smaller learning rate to avoid \n    # overshooting.\n    # lr_min = 1e-5: This is because the model should converge to stable solution with small learning rate.\n    start_lr, max_lr, min_lr = 5e-5, 8e-6 * batch_size, 1e-5\n    \n    # lr_ramp_phase_ep = 3: During this phase, the learning rate increases linearly from lr_start tolr_max.\n    # In this case, the learning rate will ramp up over 3 epochs.\n    # lr_sustain_phase = 0: During this phase, the learning rate remains constant at lr_max. in th case, \n    # the sustain phase skipped (set to 0) and the learning rate will immediately start decaying after \n    # the ramp up phase.\n    # lr_dacay = 0.75: After the ramp up and sustain phases, The learning rate will decay by this factor\n    # at each epoch., In this case, the learning rate will decay by 25% (1 - 0.75) at each epoch. \n    lr_ramp_phase_ep, lr_sustain_phase, lr_decay = 3, 0, 0.75\n    \n    # Learning rate update function(lrfn())\n    # This function calculates the learning rate for each epoch based on the mode and input parameters.\n    def lrfn(epoch):\n        # lr_ramp_phase_ep learning rate ramps up from lr_start to lr_max\n        if epoch < lr_ramp_phase_ep: lr = (max_lr - start_lr) / lr_ramp_phase_ep * epoch + start_lr\n        # lr_ramp_phase_ep + lr_sustain_phase, learning rate remains at lr_max.\n        elif epoch < lr_ramp_phase_ep + lr_sustain_phase: lr = max_lr\n        # 'exp', The learning rate decay exponentailly from lr_max to lr_min.\n        elif mode == 'exp': lr = (max_lr - min_lr) * lr_decay ** (epoch - lr_ramp_phase_ep - lr_sustain_phase) + min_lr\n        # 'step', The learning rate decays in steps from lr_max to lr_min.\n        elif mode == 'step': lr = max_lr * lr_decay ** ((epoch - lr_ramp_phase_ep - lr_sustain_phase) // 2)\n        # 'cos', The learning rate follows a cosine curve from lr_max to lr_min.\n        elif mode == 'cos':\n            # decay_epoch is the total number epochs for the decay phase.\n            # decay_index is the current epoch index in the decay phase.\n            decay_epoch, decay_index = epochs - lr_ramp_phase_ep - lr_sustain_phase + 3, epoch - lr_ramp_phase_ep - lr_sustain_phase\n            # Phase is the phase of the cosine curve, calculate as math.pi * decay_epoch / decay_index.\n            phase = math.pi * decay_index / decay_epoch\n            # The learning rate is calculated as =\n            lr = (max_lr - min_lr) * 0.5 * (1 + math.cos(phase)) + min_lr\n            \n        return lr\n    \n    # Plot lrfn function\n    plt.figure(figsize = (12, 6))\n    plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker = 'o', color = 'red')\n    plt.bar(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], color = '#33F0FF', alpha = 0.2)\n    plt.xlabel('epoch')\n    plt.ylabel('lr')\n    plt.title('LR SCHEDULE')\n    plt.show()\n    \n    return keras.callbacks.LearningRateScheduler(lrfn, verbose = False)\n    \n# Call\np_lr_s = plot_lr_schudule(config.batch_size)\np_lr_s","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:52:24.719005Z","iopub.execute_input":"2024-06-07T09:52:24.719506Z","iopub.status.idle":"2024-06-07T09:52:25.084363Z","shell.execute_reply.started":"2024-06-07T09:52:24.719461Z","shell.execute_reply":"2024-06-07T09:52:25.083215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Check Point","metadata":{}},{"cell_type":"code","source":"%%time\n# ModelCheckpoint is a callback function in keras that save models weights at certain epochs during training.\nmodel_checkpnt = keras.callbacks.ModelCheckpoint('checkpoint.weights.h5',\n                                                 # 'checkpoint.weights.h5: This is the filepath \n                                                 # where the models weights will be saved.\n                                                 monitor = 'val_auc',\n                                                 # monitor: This specifies the matric to monitor for saving the best model.\n                                                 # In this case, it's monitoring the validation AUC(area under ROC curve).\n                                                 save_best_only = True,\n                                                 # save_best_only: This mean that only the best model \n                                                 # according to the monitored metric will be saved.\n                                                 save_weights_only = True,\n                                                 # save_weights_only: This mean that only the models weights will be saved, \n                                                 # not the entire model architecture. This is usefull for saving space.\n                                                 mode = 'max')\n                                                 # This specifies the mode for saving the best model. In this case, it's\n                                                 # set to max, which mean the model with the highest monitored metric(AUC)\n                                                 # will be saved.","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:52:29.850460Z","iopub.execute_input":"2024-06-07T09:52:29.851193Z","iopub.status.idle":"2024-06-07T09:52:29.858829Z","shell.execute_reply.started":"2024-06-07T09:52:29.851151Z","shell.execute_reply":"2024-06-07T09:52:29.857682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit the model ","metadata":{}},{"cell_type":"code","source":"%%time\n# Fit the model to epochs of ten for training the model\nhistory = model.fit(train_ds,\n                    # train_ds: Input data\n                    validation_data = valid_ds,\n                    # validation_data: Data on which to evaluate the loss at any model\n                    # metrics at the end of each epochs.\n                    epochs = config.epochs,\n                    # epochs: Number of epochs to train the model\n                    callbacks = [p_lr_s, model_checkpnt],\n                    # callbacks: List of callbacks apply during training.\n                    verbose = 1)\n                    # verbose: Is recommended when not running interactively","metadata":{"execution":{"iopub.status.busy":"2024-06-07T09:52:33.415249Z","iopub.execute_input":"2024-06-07T09:52:33.416074Z","iopub.status.idle":"2024-06-07T10:56:16.089514Z","shell.execute_reply.started":"2024-06-07T09:52:33.416043Z","shell.execute_reply":"2024-06-07T10:56:16.088411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Results for AUC score and Epochs indexes","metadata":{}},{"cell_type":"code","source":"# Finds the maximum validation AUC score \nbest_auc = max(history.history['val_auc'])\n# Finds the index of the maximum AUC score and adds 1 to get the\n# epoch number(since epoch indices start from 0)\nbest_epoch = history.history['val_auc'].index(best_auc) + 1\n# Print the results \nprint('Best AUC:', best_auc)\nprint('Best Epochs:', best_epoch)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T10:59:07.328705Z","iopub.execute_input":"2024-06-07T10:59:07.329899Z","iopub.status.idle":"2024-06-07T10:59:07.336209Z","shell.execute_reply.started":"2024-06-07T10:59:07.329858Z","shell.execute_reply":"2024-06-07T10:59:07.335210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating new DataFrame ","metadata":{}},{"cell_type":"code","source":"# The glob module is a powerful tool for searchingand matching file paths\ndf_paths = glob('/kaggle/input/birdclef-2024/test_soundscapes/*.ogg')\n\n# glob is used to find all files with the .ogg extentions in the test_soundscapes directory.\n# If no files is found, it falls back to using the first 10 files from the\n# unlabeled_soundscapes directory.\nif len(df_paths) == 0:\n    df_paths = glob('/kaggle/input/birdclef-2024/unlabeled_soundscapes/*.ogg')[:10]\n\n# Creating new DataFrame filepath\ndf_tst_paths = pd.DataFrame(df_paths, columns = ['test_filepaths'])\ndf_tst_paths.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T10:59:15.730996Z","iopub.execute_input":"2024-06-07T10:59:15.731366Z","iopub.status.idle":"2024-06-07T10:59:15.899750Z","shell.execute_reply.started":"2024-06-07T10:59:15.731339Z","shell.execute_reply":"2024-06-07T10:59:15.898763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# Input meaning can handle input data with any height, width, and 3 color channels (RGB)\ninput_data = keras.layers.Input(shape = (None, None, 3))\n\n# This is the EfficientNetV2B2 backbone model, which takes the input data and processes\n#  it through several convolutional and pooling layers. The output shape is (None, None, None, 1408),\n#  indicating that it produces 1408 feature maps.\nres_backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(config.preset)\n\n#Backbone\nx = res_backbone(input_data)\n\n# This is a convolutional layers of 32 filters, taking the output from the backbone model and producing \n# 32 feature maps. \nx = keras.layers.Conv2D(32, 3, activation = 'relu')(x)\n\n# This layer performs global average pooling on the output from the convolational layer, reducing the spatial\n# dimensions to produce 1D output with 32 elements \nx = keras.layers.GlobalAveragePooling2D()(x)\n\n# This is a dense (fully connected) layers with 182 units, taking the output from the global average pooling \n# layer and producing a final output of 182 elements.\nout_data = keras.layers.Dense(config.number_c, activation = 'softmax')(x)\n\n# The model takes a input layer with shape of Input function, processes it through two hidden layers, and\n# produces output layer with shape of 182. This represent a neurel network that takes a Input\n# function dimensional input and produces a 182 dimensional output.\nmodel = keras.models.Model(inputs = input_data, outputs = out_data)\n\n# load the model weights from checkpoint\nmodel.load_weights('/kaggle/input/checkpoint-weight-h5/checkpoint.weights.h5')","metadata":{"execution":{"iopub.status.busy":"2024-06-07T10:59:24.953649Z","iopub.execute_input":"2024-06-07T10:59:24.954235Z","iopub.status.idle":"2024-06-07T10:59:31.476923Z","shell.execute_reply.started":"2024-06-07T10:59:24.954188Z","shell.execute_reply":"2024-06-07T10:59:31.475578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"# Initialize empty lists to store ID's and predictions. \nids = []\npreds = []\n\n# Build a test dataset from a list of audio file paths.\ndf_paths = df_tst_paths.test_filepaths.tolist()\ntest_ds = b_dataset(paths = df_paths, batch_size = 1)\n\n# Iterate over each auto file in the test dataset \nfor i, specs in enumerate(tqdm(iter(test_ds), desc = 'test ', total = len(df_tst_paths))):\n    # Extract the filename without extentions and with split and replace\n    file_name = df_paths[i].split('/')[-1].replace('.ogg', '')\n    # Convert the audio spectrogram to a backend-specific tensor, excluding\n    # the extra dimensions.\n    specs = keras.ops.convert_to_tensor(specs[0])\n    # np.expand_dims: This will add a singleton dimensions to the begining of the specs array,\n    # making its shape compatible with expected input shape of the model.\n    specs = np.expand_dims(specs, axis = 0)\n    # Predict bird species for all frames in the recording using the pre-trained model.\n    frame_predict = model.predict(specs, verbose = 0)\n    # Create ID's for each frame in the recording using the filename and the frame numbers.\n    ids.extend([f'{file_name}_{(frame_id + 1) * 5}' for frame_id in range(len(frame_predict))])\n    # Append to list\n    preds.append(frame_predict)\n\n# Then used np.vstack to concatenate the predictions into single array.\npreds = np.vstack(preds)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T10:59:44.050198Z","iopub.execute_input":"2024-06-07T10:59:44.051009Z","iopub.status.idle":"2024-06-07T11:00:21.270223Z","shell.execute_reply.started":"2024-06-07T10:59:44.050979Z","shell.execute_reply":"2024-06-07T11:00:21.268994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Create a new DataFrame. \npred_df = pd.DataFrame(ids, columns = ['row_id'])\n# pd.concat: Function to concatenate the pred_df DataFrame with new DataFrame created from the \n# predictions array and config.c_names column names.This ensures that the length of the keys\n# matches the length of the values.\npred_df = pd.concat([pred_df, pd.DataFrame(preds, columns = config.c_names)], axis = 1)\n# The predictions to submission.csv DataFrame.\npred_df.to_csv('submission.csv', index = False)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-07T11:00:25.011158Z","iopub.execute_input":"2024-06-07T11:00:25.012078Z","iopub.status.idle":"2024-06-07T11:00:25.046981Z","shell.execute_reply.started":"2024-06-07T11:00:25.012042Z","shell.execute_reply":"2024-06-07T11:00:25.045869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}