{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19596,"databundleVersionId":1292430,"sourceType":"competition"},{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":33246,"databundleVersionId":3221581,"sourceType":"competition"},{"sourceId":44224,"databundleVersionId":5188730,"sourceType":"competition"},{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":5195317,"sourceType":"datasetVersion","datasetId":3020983},{"sourceId":5258767,"sourceType":"datasetVersion","datasetId":3060198},{"sourceId":5258926,"sourceType":"datasetVersion","datasetId":3060292},{"sourceId":5259354,"sourceType":"datasetVersion","datasetId":3060577},{"sourceId":5259641,"sourceType":"datasetVersion","datasetId":3060750},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"\n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport numpy as np \nimport pandas as pd\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\nimport json\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-06T16:01:59.113362Z","iopub.execute_input":"2024-06-06T16:01:59.113798Z","iopub.status.idle":"2024-06-06T16:01:59.123266Z","shell.execute_reply.started":"2024-06-06T16:01:59.113765Z","shell.execute_reply":"2024-06-06T16:01:59.121754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TensorFlow:\", tf.__version__)\nprint(\"Keras:\", keras.__version__)\nprint(\"KerasCV:\", keras_cv.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T16:01:59.125521Z","iopub.execute_input":"2024-06-06T16:01:59.126009Z","iopub.status.idle":"2024-06-06T16:01:59.150218Z","shell.execute_reply.started":"2024-06-06T16:01:59.125977Z","shell.execute_reply":"2024-06-06T16:01:59.147989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = False\n    \n    verbose = 0\n    \n    training_plot = True\n    \n    device = 'TPU-VM'\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 32\n    upsample_thr = 50\n    cv_filter = True\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 20\n    preset = 'efficientnetv2_b2_imagenet'\n    fsr = False\n    num_fold = 5\n    \n    #Pretraining, neck features and final activation function\n    pretrain = 'imagenet'\n    neck_features = 0\n    final_act = 'softmax'\n    \n    # Loss function and label smoothing\n    loss = 'CCE' # BCE, CCE\n    label_smoothing = 0.05 # label smoothing\n    \n    # Learning rate, optimizer, and scheduler\n    lr = 1e-3\n    scheduler = 'cos'\n    optimizer = 'Adam' # AdamW, Adam\n    \n    # Data augmentation parameters\n    augment=True\n    \n     # Time Freq masking\n    freq_mask_prob=0.50\n    num_freq_masks=1\n    freq_mask_param=10\n    time_mask_prob=0.50\n    num_time_masks=2\n    time_mask_param=25\n\n    # Audio Augmentation Settings\n    audio_augment_prob = 0.5\n    \n    mixup_prob = 0.65\n    mixup_alpha = 0.5\n    \n    cutmix_prob = 0.65\n    cutmix_alpha = 2.5\n    \n    timeshift_prob = 0.0\n    \n    gn_prob = 0.35\n\n    # Class Labels for BirdCLEF 24\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}\n    \n    # Class Labels for BirdCLEF 21 & 22\n    class_names2 = sorted(set(os.listdir('/kaggle/input/birdclef-2021/train_short_audio/')\n                       +os.listdir('/kaggle/input/birdclef-2022/train_audio/')\n                       +os.listdir('/kaggle/input/birdsong-recognition/train_audio/')))\n    num_classes2 = len(class_names2)\n    class_labels2 = list(range(num_classes2))\n    label2name2 = dict(zip(class_labels2, class_names2))\n    name2label2 = {v:k for k,v in label2name2.items()}\n\n    # Training Settings\n    target_col = ['target']\n    tab_cols = ['filename']\n    monitor = 'auc'","metadata":{"execution":{"iopub.status.busy":"2024-06-06T16:01:59.153124Z","iopub.execute_input":"2024-06-06T16:01:59.153622Z","iopub.status.idle":"2024-06-06T16:01:59.174210Z","shell.execute_reply.started":"2024-06-06T16:01:59.153580Z","shell.execute_reply":"2024-06-06T16:01:59.172641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T16:01:59.176786Z","iopub.execute_input":"2024-06-06T16:01:59.177356Z","iopub.status.idle":"2024-06-06T16:01:59.186374Z","shell.execute_reply.started":"2024-06-06T16:01:59.177312Z","shell.execute_reply":"2024-06-06T16:01:59.185123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'\n# BASE_PATH_2023 = '/kaggle/input/birdclef-2023'\n\nBASE_PATH0 = '/kaggle/input/birdsong-recognition'\nBASE_PATH1 = '/kaggle/input/birdclef-2021'\nBASE_PATH2 = '/kaggle/input/birdclef-2022'\nBASE_PATH3 = '/kaggle/input/birdclef-2023'\nBASE_PATH4 = '/kaggle/input/xeno-canto-bird-recordings-extended-a-m-32khz-ogg'\nBASE_PATH5 = '/kaggle/input/xeno-canto-bird-recordings-extended-n-z-32khz-ogg'\nBASE_PATH6 ='/kaggle/input/cornell-birdsong-recognition-2020-n-z-32khz-ogg1'\nBASE_PATH7 ='/kaggle/input/cornell-birdsong-recognition-2020-a-m-32khz-ogg'\n\nif CFG.device==\"TPU\":\n    from kaggle_datasets import KaggleDatasets\n    GCS_PATH0 = KaggleDatasets().get_gcs_path(BASE_PATH0.split('/')[-1])\n    GCS_PATH1 = KaggleDatasets().get_gcs_path(BASE_PATH1.split('/')[-1])\n    GCS_PATH2 = KaggleDatasets().get_gcs_path(BASE_PATH2.split('/')[-1])\n    GCS_PATH3 = KaggleDatasets().get_gcs_path(BASE_PATH3.split('/')[-1])\n    GCS_PATH4 = KaggleDatasets().get_gcs_path(BASE_PATH4.split('/')[-1])\n    GCS_PATH5 = KaggleDatasets().get_gcs_path(BASE_PATH5.split('/')[-1])\n    GCS_PATH6 = KaggleDatasets().get_gcs_path(BASE_PATH6.split('/')[-1])\n    GCS_PATH7 = KaggleDatasets().get_gcs_path(BASE_PATH7.split('/')[-1])\nelse:\n    GCS_PATH0 = BASE_PATH0\n    GCS_PATH1 = BASE_PATH1\n    GCS_PATH2 = BASE_PATH2\n    GCS_PATH3 = BASE_PATH3\n    GCS_PATH4 = BASE_PATH4\n    GCS_PATH5 = BASE_PATH5\n    GCS_PATH6 = BASE_PATH6\n    GCS_PATH7 = BASE_PATH7","metadata":{"execution":{"iopub.status.busy":"2024-06-06T16:01:59.189548Z","iopub.execute_input":"2024-06-06T16:01:59.190097Z","iopub.status.idle":"2024-06-06T16:01:59.212219Z","shell.execute_reply.started":"2024-06-06T16:01:59.190054Z","shell.execute_reply":"2024-06-06T16:01:59.210167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BirdCLEF-2024\ndf_24 = pd.read_csv(f'{BASE_PATH}/train_metadata.csv')\ndf_24['filepath'] = BASE_PATH + '/train_audio/' + df_24.filename\ndf_24['target'] = df_24.primary_label.map(CFG.name2label)\ndf_24['filename'] = df_24.filepath.map(lambda x: x.split('/')[-1])\ndf_24['xc_id'] = df_24.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n\n# BirdCLEF-2023\ndf_23 = pd.read_csv(f'{BASE_PATH3}/train_metadata.csv')\ndf_23['filepath'] = GCS_PATH3 + '/train_audio/' + df_23.filename\ndf_23['target'] = df_23.primary_label.map(CFG.name2label)\ndf_23['birdclef'] = '23'\ndf_23['filename'] = df_23.filepath.map(lambda x: x.split('/')[-1])\ndf_23['xc_id'] = df_23.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n# BirdCLEF-2020\ndf_20 = pd.read_csv(f'{BASE_PATH0}/train.csv')\ndf_20['primary_label'] = df_20['ebird_code']\ndf_20['filepath'] = GCS_PATH0 + '/train_audio/' + df_20.primary_label + '/' + df_20.filename\ndf_20['scientific_name'] = df_20['sci_name']\ndf_20['common_name'] = df_20['species']\ndf_20['target'] = df_20.primary_label.map(CFG.name2label2)\ndf_20['birdclef'] = '20'\n\n#BirdCLEF-2020-Extended\ndf_20_ex1 = pd.read_csv(f'{BASE_PATH6}/train.csv')\ndf_20_ex1['filepath'] = GCS_PATH6 + '/train_audio/' + df_20_ex1.primary_label + '/' + df_20_ex1.filename\ndf_20_ex2 = pd.read_csv(f'{BASE_PATH7}/train.csv')\ndf_20_ex2['filepath'] = GCS_PATH7 + '/train_audio/' + df_20_ex1.primary_label + '/' + df_20_ex1.filename\ndf_20_ex = pd.concat([df_20_ex1, df_20_ex2], axis=0, ignore_index=True)\n\ndf_20_ex['primary_label'] = df_20_ex['ebird_code']\ndf_20_ex['scientific_name'] = df_20_ex['sci_name']\ndf_20_ex['common_name'] = df_20_ex['species']\ndf_20_ex['target'] = df_20_ex.primary_label.map(CFG.name2label2)\ndf_20_ex['birdclef'] = '20'\n\n# Xeno-Canto Extend by @vopani\ndf_xam = pd.read_csv(f'{BASE_PATH4}/train_extended.csv')\ndf_xam['filepath'] = GCS_PATH4 + '/A-M/' + df_xam.ebird_code + '/' + df_xam.filename\ndf_xnz = pd.read_csv(f'{BASE_PATH5}/train_extended.csv')\ndf_xnz['filepath'] = GCS_PATH5 + '/N-Z/' + df_xnz.ebird_code + '/' + df_xnz.filename\ndf_xc = pd.concat([df_xam, df_xnz], axis=0, ignore_index=True)\ndf_xc['primary_label'] = df_xc['ebird_code']\ndf_xc['scientific_name'] = df_xc['sci_name']\ndf_xc['common_name'] = df_xc['species']\ndf_xc['target'] = df_xc.primary_label.map(CFG.name2label2)\ndf_xc['birdclef'] = 'xc'\n\n# BirdCLEF-2021\ndf_21 = pd.read_csv(f'{BASE_PATH1}/train_metadata.csv')\ndf_21['filepath'] = GCS_PATH1 + '/train_short_audio/' + df_21.primary_label + '/' + df_21.filename\ndf_21['target'] = df_21.primary_label.map(CFG.name2label2)\ndf_21['birdclef'] = '21'\ncorrupt_paths = ['/kaggle/input/birdclef-2021/train_short_audio/houwre/XC590621.ogg',\n                 '/kaggle/input/birdclef-2021/train_short_audio/cogdov/XC579430.ogg']\ndf_21 = df_21[~df_21.filepath.isin(corrupt_paths)] # remove all zero audios\n\n# BirdCLEF-2022\ndf_22 = pd.read_csv(f'{BASE_PATH2}/train_metadata.csv')\ndf_22['filepath'] = GCS_PATH2 + '/train_audio/' + df_22.filename\ndf_22['target'] = df_22.primary_label.map(CFG.name2label2)\ndf_22['birdclef'] = '22'\n\n# Merge 2021 and 2022 for pretraining\ndf_pre = pd.concat([df_20, df_20_ex, df_21, df_22,df_23, df_xc], axis=0, ignore_index=True)\ndf_pre['filename'] = df_pre.filepath.map(lambda x: x.split('/')[-1])\ndf_pre['xc_id'] = df_pre.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\nnodup_idx = df_pre[['xc_id','primary_label','author']].drop_duplicates().index\ndf_pre = df_pre.loc[nodup_idx].reset_index(drop=True)\n\n# # Remove duplicates\ndf_pre = df_pre[~df_pre.xc_id.isin(df_24.xc_id)].reset_index(drop=True)\ncorrupt_mp3s = json.load(open('/kaggle/input/birdclef-corrupt-mp3-files-ds/corrupt_mp3_files.json','r'))\ndf_pre = df_pre[~df_pre.filepath.isin(corrupt_mp3s)]\ndf_pre = df_pre[['filename','filepath','primary_label','secondary_labels',\n                 'rating','author','file_type','xc_id','scientific_name',\n                'common_name','target','birdclef','bird_seen']]\n# Display rows\nprint(\"# Samples for Pre-Training: {:,}\".format(len(df_pre)))\ndf_pre.head(2).style.set_caption(\"Pre-Training Data\").set_table_styles([{\n    'selector': 'caption',\n    'props': [\n        ('color', 'blue'),\n        ('font-size', '16px')\n    ]\n}])\n\n# Show distribution\nplt.figure(figsize=(8, 4))\ndf_pre.birdclef.value_counts().plot.bar(color=[cmap(0.0),cmap(0.25), cmap(0.65), cmap(0.9)])\nplt.xlabel(\"Dataset\")\nplt.ylabel(\"Count\")\nplt.title(\"Dataset distribution for Pre-Training\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T16:22:28.227276Z","iopub.execute_input":"2024-06-06T16:22:28.227731Z","iopub.status.idle":"2024-06-06T16:22:33.236652Z","shell.execute_reply.started":"2024-06-06T16:22:28.227674Z","shell.execute_reply":"2024-06-06T16:22:33.235213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display rwos\ndf_2024.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.583001Z","iopub.status.idle":"2024-06-06T15:08:53.583540Z","shell.execute_reply.started":"2024-06-06T15:08:53.583254Z","shell.execute_reply":"2024-06-06T15:08:53.583277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_2023 = pd.read_csv(f'{BASE_PATH_2023}/train_metadata.csv')\ndf_2023['filepath'] = BASE_PATH_2023 + '/train_audio/' + df_2023.filename\ndf_2023['target'] = df_2023.primary_label.map(CFG.name2label)\ndf_2023['filename'] = df_2023.filepath.map(lambda x: x.split('/')[-1])\ndf_2023['xc_id'] = df_2023.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\ndf_2023.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.585011Z","iopub.status.idle":"2024-06-06T15:08:53.585515Z","shell.execute_reply.started":"2024-06-06T15:08:53.585247Z","shell.execute_reply":"2024-06-06T15:08:53.585267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_primary_labels_2024 = set(df_2024['primary_label'].unique())\nunique_primary_labels_2023 = set(df_2023['primary_label'].unique())\n\nintersection_labels = unique_primary_labels_2024.intersection(unique_primary_labels_2023)\n\nprint(intersection_labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.587099Z","iopub.status.idle":"2024-06-06T15:08:53.587503Z","shell.execute_reply.started":"2024-06-06T15:08:53.587307Z","shell.execute_reply":"2024-06-06T15:08:53.587324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_df_2023 = df_2023[df_2023[\"primary_label\"].isin(intersection_labels)]\n\ndf = pd.concat([df_2024, filtered_df_2023], ignore_index = True)\n\ndf['target'] = df.primary_label.map(CFG.name2label).astype(int)\n\nprint(\"Number of 'None' values in target column:\", df['target'].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.589199Z","iopub.status.idle":"2024-06-06T15:08:53.589568Z","shell.execute_reply.started":"2024-06-06T15:08:53.589381Z","shell.execute_reply":"2024-06-06T15:08:53.589396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df) - len(df_2023))","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.591503Z","iopub.status.idle":"2024-06-06T15:08:53.592063Z","shell.execute_reply.started":"2024-06-06T15:08:53.591744Z","shell.execute_reply":"2024-06-06T15:08:53.591764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.593293Z","iopub.status.idle":"2024-06-06T15:08:53.593843Z","shell.execute_reply.started":"2024-06-06T15:08:53.593545Z","shell.execute_reply":"2024-06-06T15:08:53.593567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop_duplicates(subset=['xc_id','author','primary_label'])","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.595376Z","iopub.status.idle":"2024-06-06T15:08:53.595726Z","shell.execute_reply.started":"2024-06-06T15:08:53.595553Z","shell.execute_reply":"2024-06-06T15:08:53.595568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.597019Z","iopub.status.idle":"2024-06-06T15:08:53.597368Z","shell.execute_reply.started":"2024-06-06T15:08:53.597191Z","shell.execute_reply":"2024-06-06T15:08:53.597207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"species_counts = df['primary_label'].value_counts()\n\nspecies_to_include = species_counts[species_counts >= 10].index\n\ndf = df[df['primary_label'].isin(species_to_include)]","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.598727Z","iopub.status.idle":"2024-06-06T15:08:53.599102Z","shell.execute_reply.started":"2024-06-06T15:08:53.598925Z","shell.execute_reply":"2024-06-06T15:08:53.598940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.600562Z","iopub.status.idle":"2024-06-06T15:08:53.600934Z","shell.execute_reply.started":"2024-06-06T15:08:53.600728Z","shell.execute_reply":"2024-06-06T15:08:53.600742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef visualize_label_distribution(df):\n    \"\"\"\n    Visualizes the distribution of primary labels in the DataFrame.\n    \n    Args:\n    - df (pd.DataFrame): DataFrame containing the data with 'primary_label' column.\n    \n    Returns:\n    - None\n    \"\"\"\n    # Count occurrences of each primary label\n    label_counts = df['primary_label'].value_counts()\n    \n    # Plotting\n    plt.figure(figsize=(12, 6))\n    label_counts.plot(kind='bar')\n    plt.title('Distribution of Primary Labels')\n    plt.xlabel('Primary Labels')\n    plt.ylabel('Frequency')\n    plt.xticks(rotation=45)\n    plt.show()\n\n# Usage example:\nvisualize_label_distribution(df_24)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T16:23:00.632239Z","iopub.execute_input":"2024-06-06T16:23:00.632713Z","iopub.status.idle":"2024-06-06T16:23:02.936372Z","shell.execute_reply.started":"2024-06-06T16:23:00.632677Z","shell.execute_reply":"2024-06-06T16:23:02.934878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio(filepath):\n    audio, sr = librosa.load(filepath)\n    return audio, sr\n\ndef get_spectrogram(audio):\n    spec = librosa.feature.melspectrogram(y=audio, \n                                   sr=CFG.sample_rate, \n                                   n_mels=256,\n                                   n_fft=2048,\n                                   hop_length=512,\n                                   fmax=CFG.fmax,\n                                   fmin=CFG.fmin,\n                                   )\n    spec = librosa.power_to_db(spec, ref=1.0)\n    min_ = spec.min()\n    max_ = spec.max()\n    if max_ != min_:\n        spec = (spec - min_)/(max_ - min_)\n    return spec\n\ndef display_audio(row):\n    # Caption for viz\n    caption = f'Id: {row.filename} | Name: {row.common_name} | Sci.Name: {row.scientific_name} | Rating: {row.rating}'\n    # Read audio file\n    audio, sr = load_audio(row.filepath)\n    # Keep fixed length audio\n    audio = audio[:CFG.audio_len]\n    # Spectrogram from audio\n    spec = get_spectrogram(audio)\n    # Display audio\n    print(\"# Audio:\")\n    display(ipd.Audio(audio, rate=CFG.sample_rate))\n    print('# Visualization:')\n    fig, ax = plt.subplots(2, 1, figsize=(12, 2*3), sharex=True, tight_layout=True)\n    fig.suptitle(caption)\n    # Waveplot\n    lid.waveshow(audio,\n                 sr=CFG.sample_rate,\n                 ax=ax[0],\n                 color= cmap(0.1))\n    # Specplot\n    lid.specshow(spec, \n                 sr = CFG.sample_rate, \n                 hop_length=512,\n                 n_fft=2048,\n                 fmin=CFG.fmin,\n                 fmax=CFG.fmax,\n                 x_axis = 'time', \n                 y_axis = 'mel',\n                 cmap = 'coolwarm',\n                 ax=ax[1])\n    ax[0].set_xlabel('');\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.604429Z","iopub.status.idle":"2024-06-06T15:08:53.604776Z","shell.execute_reply.started":"2024-06-06T15:08:53.604615Z","shell.execute_reply":"2024-06-06T15:08:53.604629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = df.iloc[35]\n\n# Display audio\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.606010Z","iopub.status.idle":"2024-06-06T15:08:53.606408Z","shell.execute_reply.started":"2024-06-06T15:08:53.606195Z","shell.execute_reply":"2024-06-06T15:08:53.606237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import required packages\nfrom sklearn.model_selection import train_test_split\n\ntrain_df, valid_df = train_test_split(df, test_size=0.2)\n\nprint(f\"Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.608620Z","iopub.status.idle":"2024-06-06T15:08:53.609040Z","shell.execute_reply.started":"2024-06-06T15:08:53.608813Z","shell.execute_reply":"2024-06-06T15:08:53.608850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decodes Audio\ndef build_decoder(with_labels=True, dim=1024):\n    def get_audio(filepath):\n        file_bytes = tf.io.read_file(filepath)\n        audio = tfio.audio.decode_vorbis(file_bytes)  # decode .ogg file\n        audio = tf.cast(audio, tf.float32)\n        if tf.shape(audio)[1] > 1:  # stereo -> mono\n            audio = audio[..., 0:1]\n        audio = tf.squeeze(audio, axis=-1)\n        return audio\n\n    def crop_or_pad(audio, target_len, pad_mode=\"constant\"):\n        audio_len = tf.shape(audio)[0]\n        diff_len = abs(\n            target_len - audio_len\n        )  # find difference between target and audio length\n        if audio_len < target_len:  # do padding if audio length is shorter\n            pad1 = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            pad2 = diff_len - pad1\n            audio = tf.pad(audio, paddings=[[pad1, pad2]], mode=pad_mode)\n        elif audio_len > target_len:  # do cropping if audio length is larger\n            idx = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            audio = audio[idx : (idx + target_len)]\n        return tf.reshape(audio, [target_len])\n\n    def apply_preproc(spec):\n        # Standardize\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        # Normalize using Min-Max\n        min_val = tf.math.reduce_min(spec)\n        max_val = tf.math.reduce_max(spec)\n        spec = tf.where(\n            tf.math.equal(max_val - min_val, 0),\n            spec - min_val,\n            (spec - min_val) / (max_val - min_val),\n        )\n        return spec\n\n    def get_target(target):\n        target = tf.reshape(target, [1])\n        target = tf.cast(tf.one_hot(target, CFG.num_classes), tf.float32)\n        target = tf.reshape(target, [CFG.num_classes])\n        return target\n\n    def decode(path):\n        # Load audio file\n        audio = get_audio(path)\n        # Crop or pad audio to keep a fixed length\n        audio = crop_or_pad(audio, dim)\n        # Audio to Spectrogram\n        spec = keras.layers.MelSpectrogram(\n            num_mel_bins=CFG.img_size[0],\n            fft_length=CFG.nfft,\n            sequence_stride=CFG.hop_length,\n            sampling_rate=CFG.sample_rate,\n        )(audio)\n        # Apply normalization and standardization\n        spec = apply_preproc(spec)\n        # Spectrogram to 3 channel image (for imagenet)\n        spec = tf.tile(spec[..., None], [1, 1, 3])\n        spec = tf.reshape(spec, [*CFG.img_size, 3])\n        return spec\n\n    def decode_with_labels(path, label):\n        label = get_target(label)\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.611007Z","iopub.status.idle":"2024-06-06T15:08:53.611547Z","shell.execute_reply.started":"2024-06-06T15:08:53.611260Z","shell.execute_reply":"2024-06-06T15:08:53.611282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_augmenter():\n    augmenters = [\n        keras_cv.layers.MixUp(alpha=0.4),\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                     width_factor=(0.06, 0.12)), # time-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                     width_factor=(1.0, 1.0)), # freq-masking\n    ]\n    \n    def augment(img, label):\n        data = {\"images\":img, \"labels\":label}\n        for augmenter in augmenters:\n            if tf.random.uniform([]) < 0.35:\n                data = augmenter(data, training=True)\n        return data[\"images\"], data[\"labels\"]\n    \n    return augment\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.612957Z","iopub.status.idle":"2024-06-06T15:08:53.613480Z","shell.execute_reply.started":"2024-06-06T15:08:53.613203Z","shell.execute_reply":"2024-06-06T15:08:53.613225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_dataset(paths, labels=None, batch_size=32, \n                  decode_fn=None, augment_fn=None, cache=True,\n                  augment=False, shuffle=2048):\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None, dim=CFG.audio_len)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter()\n        \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths,) if labels is None else (paths, labels)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache() if cache else ds\n    if shuffle:\n        opt = tf.data.Options()\n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=True)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.614881Z","iopub.status.idle":"2024-06-06T15:08:53.615391Z","shell.execute_reply.started":"2024-06-06T15:08:53.615119Z","shell.execute_reply":"2024-06-06T15:08:53.615141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train\ntrain_paths = train_df.filepath.values\ntrain_labels = train_df.target.values\ntrain_ds = build_dataset(train_paths, train_labels, batch_size=CFG.batch_size,\n                         shuffle=True, augment=CFG.augment)\n\n# Valid\nvalid_paths = valid_df.filepath.values\nvalid_labels = valid_df.target.values\nvalid_ds = build_dataset(valid_paths, valid_labels, batch_size=CFG.batch_size,\n                         shuffle=False, augment=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.617629Z","iopub.status.idle":"2024-06-06T15:08:53.618182Z","shell.execute_reply.started":"2024-06-06T15:08:53.617901Z","shell.execute_reply":"2024-06-06T15:08:53.617925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%debug","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.619850Z","iopub.status.idle":"2024-06-06T15:08:53.620374Z","shell.execute_reply.started":"2024-06-06T15:08:53.620102Z","shell.execute_reply":"2024-06-06T15:08:53.620123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an input layer for the model\ninp = keras.layers.Input(shape=(None, None, 3))\n# Pretrained backbone\nbackbone = keras_cv.models.EfficientNetV2Backbone.from_preset(\n    CFG.preset,\n)\nout = keras_cv.models.ImageClassifier(\n    backbone=backbone,\n    num_classes=CFG.num_classes,\n    name=\"classifier\"\n)(inp)\n# Build model\nmodel = keras.models.Model(inputs=inp, outputs=out)\n# Compile model with optimizer, loss and metrics\nmodel.compile(optimizer=\"adam\",\n              loss=keras.losses.CategoricalCrossentropy(label_smoothing=0.02),\n              metrics=[keras.metrics.AUC(name='auc')],\n             )\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.622002Z","iopub.status.idle":"2024-06-06T15:08:53.622538Z","shell.execute_reply.started":"2024-06-06T15:08:53.622248Z","shell.execute_reply":"2024-06-06T15:08:53.622270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\ndef get_lr_callback(batch_size=8, mode='cos', epochs=10, plot=False):\n    lr_start, lr_max, lr_min = 5e-5, 8e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        elif mode == 'exp': lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n        elif mode == 'step': lr = lr_max * lr_decay**((epoch - lr_ramp_ep - lr_sus_ep) // 2)\n        elif mode == 'cos':\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        return lr\n\n    if plot:  # Plot lr curve if plot is True\n        plt.figure(figsize=(10, 5))\n        plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('lr')\n        plt.title('LR Scheduler')\n        plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  # Create lr callback","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.624448Z","iopub.status.idle":"2024-06-06T15:08:53.624884Z","shell.execute_reply.started":"2024-06-06T15:08:53.624655Z","shell.execute_reply":"2024-06-06T15:08:53.624672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_cb = get_lr_callback(CFG.batch_size, plot=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.626179Z","iopub.status.idle":"2024-06-06T15:08:53.626542Z","shell.execute_reply.started":"2024-06-06T15:08:53.626363Z","shell.execute_reply":"2024-06-06T15:08:53.626378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.weights.h5\",\n                                         monitor='val_auc',\n                                         save_best_only=True,\n                                         save_weights_only=True,\n                                         mode='max')","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.627767Z","iopub.status.idle":"2024-06-06T15:08:53.628213Z","shell.execute_reply.started":"2024-06-06T15:08:53.628002Z","shell.execute_reply":"2024-06-06T15:08:53.628019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    validation_data=valid_ds, \n    epochs=CFG.epochs,\n    callbacks=[lr_cb, ckpt_cb], \n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.629417Z","iopub.status.idle":"2024-06-06T15:08:53.629771Z","shell.execute_reply.started":"2024-06-06T15:08:53.629589Z","shell.execute_reply":"2024-06-06T15:08:53.629604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_epoch = np.argmax(history.history[\"val_auc\"])\nbest_score = history.history[\"val_auc\"][best_epoch]\nprint('>>> Best AUC: ', best_score)\nprint('>>> Best Epoch: ', best_epoch+1)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T15:08:53.631479Z","iopub.status.idle":"2024-06-06T15:08:53.631859Z","shell.execute_reply.started":"2024-06-06T15:08:53.631652Z","shell.execute_reply":"2024-06-06T15:08:53.631666Z"},"trusted":true},"execution_count":null,"outputs":[]}]}