{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":12321821,"sourceType":"datasetVersion","datasetId":7762955},{"sourceId":6127,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":4598,"modelId":2797},{"sourceId":453041,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":367517,"modelId":388413},{"sourceId":454401,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":368711,"modelId":388413}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"colab":{"provenance":[]},"widgets":{"application/vnd.jupyter.widget-state+json":{"158442b3b5544bdd9dfa363ad936b433":{"model_module":"@jupyter-widgets/controls","model_name":"VBoxModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"VBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"VBoxView","box_style":"","children":["IPY_MODEL_0e3e97cb90644ff6b9fcb85b578e201e"],"layout":"IPY_MODEL_2d18f660324c48cbadb08c0d3d02770a"}},"5eb9aa4d7a8b4c5d8a1e00720465b0ed":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_1012c9527a7a439eaa12da1c16abe416","placeholder":"​","style":"IPY_MODEL_e4850ee3e55a4e5f8b34cae3ffc0bb5e","value":"<center> <img\nsrc=https://www.kaggle.com/static/images/site-logo.png\nalt='Kaggle'> <br> Create an API token from <a\nhref=\"https://www.kaggle.com/settings/account\" target=\"_blank\">your Kaggle\nsettings page</a> and paste it below along with your Kaggle username. <br> </center>"}},"41cb1171a5484c7bb6f73964f68f4fa9":{"model_module":"@jupyter-widgets/controls","model_name":"TextModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"TextModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"TextView","continuous_update":true,"description":"Username:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_ab3379c0bd134f3999db8c86805621f9","placeholder":"​","style":"IPY_MODEL_27db13985c2d4fdaa39ada7feea14ae1","value":"luchozm22"}},"8ae80bc4fc0e4e86ad26e0d39b84ad45":{"model_module":"@jupyter-widgets/controls","model_name":"PasswordModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"PasswordModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"PasswordView","continuous_update":true,"description":"Token:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_c46f4bd386814a8f8622803e24e5ea47","placeholder":"​","style":"IPY_MODEL_92fd946ccac643cb9dc9f001b4b9c719","value":""}},"2097097c515b4ce9bc9d879d37e2fdee":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ButtonView","button_style":"","description":"Login","disabled":false,"icon":"","layout":"IPY_MODEL_9ba35d5de0574b66a432bc446cc38a17","style":"IPY_MODEL_4c32a37cce6f46de8ff4533cd5df04e2","tooltip":""}},"3328744091ee482fb72064054762b68a":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_0f25e50b95204ceca048ff2cb6f8bf3f","placeholder":"​","style":"IPY_MODEL_313a64b29f474c9391754f59d9f4ed80","value":"\n<b>Thank You</b></center>"}},"2d18f660324c48cbadb08c0d3d02770a":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":"center","align_self":null,"border":null,"bottom":null,"display":"flex","flex":null,"flex_flow":"column","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"50%"}},"1012c9527a7a439eaa12da1c16abe416":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e4850ee3e55a4e5f8b34cae3ffc0bb5e":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"ab3379c0bd134f3999db8c86805621f9":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"27db13985c2d4fdaa39ada7feea14ae1":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"c46f4bd386814a8f8622803e24e5ea47":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"92fd946ccac643cb9dc9f001b4b9c719":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"9ba35d5de0574b66a432bc446cc38a17":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4c32a37cce6f46de8ff4533cd5df04e2":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","button_color":null,"font_weight":""}},"0f25e50b95204ceca048ff2cb6f8bf3f":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"313a64b29f474c9391754f59d9f4ed80":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"3924bca77977485a996526049e3165f0":{"model_module":"@jupyter-widgets/controls","model_name":"LabelModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"LabelModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"LabelView","description":"","description_tooltip":null,"layout":"IPY_MODEL_a45e52a0e27e42faac87e9df318e2d4b","placeholder":"​","style":"IPY_MODEL_2cf99cd0e8d34851adc1dbcba8f251db","value":"Connecting..."}},"a45e52a0e27e42faac87e9df318e2d4b":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2cf99cd0e8d34851adc1dbcba8f251db":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"0e3e97cb90644ff6b9fcb85b578e201e":{"model_module":"@jupyter-widgets/controls","model_name":"LabelModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"LabelModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"LabelView","description":"","description_tooltip":null,"layout":"IPY_MODEL_756d4fd9cfce4863ba735c78df5b8e6c","placeholder":"​","style":"IPY_MODEL_b42cd92a62a14fc5ae8ecfba8de2c8d7","value":"Kaggle credentials successfully validated."}},"756d4fd9cfce4863ba735c78df5b8e6c":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b42cd92a62a14fc5ae8ecfba8de2c8d7":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{"id":"-b7-wH2C1qMT"}},{"cell_type":"code","source":"!pip install librosa","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.346731Z","iopub.status.idle":"2025-12-04T15:00:53.346945Z","shell.execute_reply.started":"2025-12-04T15:00:53.346844Z","shell.execute_reply":"2025-12-04T15:00:53.346854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\n\nimport numpy as np \nimport pandas as pd\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"id":"NPD-GRO51qMU","outputId":"aaa6b4b4-d9c4-4f2d-e37f-bd6f26489b1e","execution":{"iopub.status.busy":"2025-12-04T15:00:53.348483Z","iopub.status.idle":"2025-12-04T15:00:53.348781Z","shell.execute_reply.started":"2025-12-04T15:00:53.348634Z","shell.execute_reply":"2025-12-04T15:00:53.348647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FILE GLOBALS","metadata":{"id":"kmin16Sp1qMV"}},{"cell_type":"code","source":"INPUT_FOLDER = '/kaggle/input'\nMAIN_FOLDER  = os.path.join(INPUT_FOLDER, 'birdclef-2024')\nTRAIN_AUDIO = os.path.join(MAIN_FOLDER,'train_audio')\nTRAIN_CSV = os.path.join(MAIN_FOLDER, 'train_metadata.csv')\nTAXONOMY_CSV = os.path.join(MAIN_FOLDER, 'eBird_Taxonomy_v2021.csv')\nUNLABELED_AUDIO = os.path.join(MAIN_FOLDER, 'unlabeled_soundscapes')","metadata":{"trusted":true,"id":"mOx6QtWi1qMW","execution":{"iopub.status.busy":"2025-12-04T15:00:53.349246Z","iopub.status.idle":"2025-12-04T15:00:53.349552Z","shell.execute_reply.started":"2025-12-04T15:00:53.349383Z","shell.execute_reply":"2025-12-04T15:00:53.349395Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CONFIGURATIONS","metadata":{}},{"cell_type":"code","source":"class CFG:\n    sample_rate = 32000\n\n    img_size = [128, 384]\n    batch_size = 64\n    \n    duration = 15\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n\n    # STFT parameters\n    nfft = 2048\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 40\n    fmax = 15000\n\n    preset = 'efficientnetv2_b2_imagenet'\n    epochs = 15\n\n    class_names = sorted(os.listdir(TRAIN_AUDIO))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.350573Z","iopub.status.idle":"2025-12-04T15:00:53.35086Z","shell.execute_reply.started":"2025-12-04T15:00:53.350713Z","shell.execute_reply":"2025-12-04T15:00:53.350727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)\ndf['filepath'] = df.filename.map(lambda x: os.path.join(TRAIN_AUDIO, x))\ndrop_columns = ['type', 'latitude', 'longitude', 'author', 'license', 'url', 'secondary_labels']\ndf = df.drop(columns=drop_columns)\ndf['target'] = df.primary_label.map(CFG.name2label)\n","metadata":{"trusted":true,"id":"KqO90hPS1qMW","execution":{"iopub.status.busy":"2025-12-04T15:00:53.352493Z","iopub.status.idle":"2025-12-04T15:00:53.352743Z","shell.execute_reply.started":"2025-12-04T15:00:53.352638Z","shell.execute_reply":"2025-12-04T15:00:53.35265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"id":"i9niBkS5ZSgm","trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.353551Z","iopub.status.idle":"2025-12-04T15:00:53.353788Z","shell.execute_reply.started":"2025-12-04T15:00:53.353672Z","shell.execute_reply":"2025-12-04T15:00:53.353685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy = pd.read_csv(TAXONOMY_CSV)\ntaxonomy = taxonomy.set_index('SPECIES_CODE')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.354851Z","iopub.status.idle":"2025-12-04T15:00:53.355067Z","shell.execute_reply.started":"2025-12-04T15:00:53.354967Z","shell.execute_reply":"2025-12-04T15:00:53.354977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_values = df[['scientific_name', 'primary_label']].drop_duplicates()\nunique_values.sample(frac=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.356316Z","iopub.status.idle":"2025-12-04T15:00:53.356569Z","shell.execute_reply.started":"2025-12-04T15:00:53.356451Z","shell.execute_reply":"2025-12-04T15:00:53.356465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"def load_ogg(path, target_sr):\n    # Load audio file at the target sample rate\n    audio, _ = librosa.load(path, sr=target_sr, mono=True)\n    return audio.astype('float32')\n\ndef crop_or_pad(audio, audio_len):\n    # Ensure audio is exactly audio_len in samples - Crop if too long\n    if len(audio) > audio_len:\n        audio = audio[:audio_len]  \n    else:\n        pad_width = audio_len - len(audio)\n        audio = np.pad(audio, (0, pad_width), mode='constant')  # Pad if too short\n    return audio\n\ndef get_spectrogram(audio):\n    audio_tensor = tf.expand_dims(audio, axis=0) \n    \n    # Create the MelSpectrogram layer\n    mel_spec_layer = tf.keras.layers.MelSpectrogram(\n        fft_length=CFG.nfft,\n        sequence_stride=CFG.hop_length,\n        sampling_rate=CFG.sample_rate,\n        num_mel_bins=CFG.img_size[0],\n        min_freq=CFG.fmin,\n        max_freq=CFG.fmax,\n    )\n\n    spec = mel_spec_layer(audio_tensor) \n    return spec  \n\ndef display_audio(row):\n    path = row.filepath\n\n    display(ipd.Audio(path))\n    \n    caption = f'file: {row.filename}, Name: {row.common_name}, Scientific Name: {row.scientific_name}'\n    fig, ax = plt.subplots(2, 1, figsize=(12, 6), sharex=True, tight_layout=True)  \n    fig.suptitle(caption)\n\n    # Load and pad/crop audio to fixed length\n    audio = load_ogg(path, CFG.sample_rate)\n    audio = crop_or_pad(audio, CFG.audio_len)\n\n    # Plot waveform\n    lid.waveshow(audio, sr=CFG.sample_rate, ax=ax[0])  # Added sr= to ensure time axis is correct\n    ax[0].set_title('Waveform')\n\n    # Plot spectrogram\n    spec = get_spectrogram(audio)\n    spec_plot = tf.squeeze(spec, axis=0)  \n    print(f'spec shape (after squeeze for plotting): {spec_plot.shape}')\n\n    lid.specshow(spec_plot.numpy(), sr=CFG.sample_rate, hop_length=CFG.hop_length, x_axis='time', y_axis='mel', ax=ax[1])  #\n\n    plt.show()  \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.534147Z","iopub.execute_input":"2025-12-04T15:00:53.534449Z","iopub.status.idle":"2025-12-04T15:00:53.542888Z","shell.execute_reply.started":"2025-12-04T15:00:53.53441Z","shell.execute_reply":"2025-12-04T15:00:53.542151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_audio(df.iloc[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.544064Z","iopub.execute_input":"2025-12-04T15:00:53.544611Z","iopub.status.idle":"2025-12-04T15:00:53.563617Z","shell.execute_reply.started":"2025-12-04T15:00:53.544594Z","shell.execute_reply":"2025-12-04T15:00:53.561889Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load and preprocess function\n\nThe following code will decode the raw audio from .ogg file and also decode the spectrogram from the audio file. Additionally, we will apply Z-Score standardization and Min-Max normalization to ensure consistent inputs to the model.","metadata":{"id":"hMT-QOUe1qMX"}},{"cell_type":"code","source":"def load_ogg_tf(path, target_sr=CFG.sample_rate, audio_len=CFG.audio_len):\n    def _load_ogg_np(filepath_tensor, sr_val):\n        filepath = filepath_tensor.numpy().decode('utf-8') \n        audio, _ = librosa.load(filepath, sr=sr_val, mono=True)\n        return audio.astype(np.float32)\n    \n    \n    audio_np = tf.py_function(func=_load_ogg_np, \n                              inp=[path, target_sr], \n                              Tout=tf.float32)\n    \n    current_len = tf.shape(audio_np)[0]\n    audio_np = tf.cond(current_len > audio_len,\n                    lambda: audio_np[:audio_len],\n                    lambda: tf.pad(audio_np, [[0, audio_len - current_len]], mode='CONSTANT'))\n    audio_np.set_shape([audio_len])\n    return audio_np\n\ndef standarize_spec(spec):\n    # Z-Score Standarization\n    mean = tf.math.reduce_mean(spec, axis=[0, 1], keepdims=True)\n    std = tf.math.reduce_std(spec, axis=[0, 1], keepdims=True)\n    \n    # Standardize with numerical stability\n    spec = (spec - mean) / (std + 1e-6)\n    return spec\n    \ndef get_spectrogram_tf(audio):\n    audio_tensor = tf.expand_dims(audio, axis=0) \n\n    \n    mel_spec_layer = tf.keras.layers.MelSpectrogram(\n        fft_length=CFG.nfft,\n        sequence_stride=CFG.hop_length,\n        sampling_rate=CFG.sample_rate,\n        num_mel_bins=CFG.img_size[0], \n        min_freq=CFG.fmin,\n        max_freq=CFG.fmax\n    )\n\n    spec = mel_spec_layer(audio_tensor)\n    spec = tf.transpose(spec, perm=[1, 2, 0]) \n\n    return standarize_spec(spec)\n\n\ndef get_target(target):\n    target = tf.reshape(target, [1])\n    target = tf.cast(tf.one_hot(target, CFG.num_classes), tf.float32)\n    target = tf.reshape(target, [CFG.num_classes])\n    return target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.563954Z","iopub.status.idle":"2025-12-04T15:00:53.564165Z","shell.execute_reply.started":"2025-12-04T15:00:53.564068Z","shell.execute_reply":"2025-12-04T15:00:53.564077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess(path, label):\n\n    audio = load_ogg_tf(path, CFG.sample_rate)\n    spectrogram = get_spectrogram_tf(audio)\n    spectrogram = standarize_spec(spectrogram)\n    spec_rgb = tf.image.grayscale_to_rgb(spectrogram)\n    label = get_target(label)\n    return spec_rgb, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.565575Z","iopub.status.idle":"2025-12-04T15:00:53.56595Z","shell.execute_reply.started":"2025-12-04T15:00:53.565771Z","shell.execute_reply":"2025-12-04T15:00:53.565782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Augmentation\nFollowing code will apply augmentations to spectrogram data. In this notebook, we will use MixUp, CutOut ","metadata":{}},{"cell_type":"code","source":"def augment_pre_batch(spec, label):\n    \"\"\"Augmentations that can be applied to single spectrograms.\"\"\"\n    augmenters = [\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0), \n                                      width_factor=(0.06, 0.12)), # Time-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1), \n                                      width_factor=(1.0, 1.0)),   # Frequency-masking\n    ]\n    \n\n    for augmenter in augmenters:\n        if tf.random.uniform([]) < 0.5: # Apply augmentations randomly\n            spec = augmenter(spec)\n        \n    return spec, label\n\ndef augment_post_batch(specs, labels):\n    \"\"\"Augmentations that are applied to a batch of spectrograms.\"\"\"\n    augmenter = keras_cv.layers.MixUp(alpha=0.4)\n    data = {\"images\": specs, \"labels\": labels} \n    data = augmenter(data, training=True)\n    return data[\"images\"], data[\"labels\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.567273Z","iopub.status.idle":"2025-12-04T15:00:53.567666Z","shell.execute_reply.started":"2025-12-04T15:00:53.567497Z","shell.execute_reply":"2025-12-04T15:00:53.567519Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Pipeline","metadata":{}},{"cell_type":"code","source":"def build_dataset(paths, labels, batch_size=CFG.batch_size, \n                  is_train=True, augment=False):\n    \n    slices = (paths, labels)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    \n    # Preprocess each sample\n    ds = ds.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)\n    \n    if is_train and augment:\n        # Apply pre-batch augmentations\n        ds = ds.map(augment_pre_batch, num_parallel_calls=tf.data.AUTOTUNE)\n\n    # Shuffle and batch the dataset\n    if is_train:\n        ds = ds.shuffle(2048)\n    ds = ds.batch(batch_size)\n\n    if is_train and augment:\n        # Apply post-batch augmentations (MixUp)\n        ds = ds.map(augment_post_batch, num_parallel_calls=tf.data.AUTOTUNE)\n\n    # Prefetch for performance\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    \n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.568831Z","iopub.status.idle":"2025-12-04T15:00:53.56908Z","shell.execute_reply.started":"2025-12-04T15:00:53.568962Z","shell.execute_reply":"2025-12-04T15:00:53.568985Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train and Test Split","metadata":{}},{"cell_type":"code","source":"train_df, test_df = train_test_split(df, test_size=0.2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.570025Z","iopub.status.idle":"2025-12-04T15:00:53.570253Z","shell.execute_reply.started":"2025-12-04T15:00:53.570121Z","shell.execute_reply":"2025-12-04T15:00:53.570131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds = build_dataset(train_df.filepath.values, train_df.target.values, is_train=True, augment=True)\ntest_ds = build_dataset(test_df.filepath.values, test_df.target.values, is_train=False, augment=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.570662Z","iopub.status.idle":"2025-12-04T15:00:53.570896Z","shell.execute_reply.started":"2025-12-04T15:00:53.570795Z","shell.execute_reply":"2025-12-04T15:00:53.570806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Show Data Set First Value","metadata":{}},{"cell_type":"code","source":"def plot_batch(batch, row=3, col=3, label2name=None):\n\n    if isinstance(batch, tuple) or isinstance(batch, list):\n        specs, tars = batch\n    else:\n        specs = batch\n        tars = None\n    plt.figure(figsize=(col*5, row*3))\n    # print(specs.shape)\n    # print(specs)\n    for idx in range(row*col):\n        ax = plt.subplot(row, col, idx+1)\n        \n        lid.specshow(np.array(specs[idx, ..., 0]), \n                     n_fft=CFG.nfft, \n                     hop_length=CFG.hop_length, \n                     sr=CFG.sample_rate,\n                     x_axis='time',\n                     y_axis='mel',\n                     cmap='coolwarm')\n        if tars is not None:\n            label = tars[idx].numpy().argmax()\n            name = label2name[label]\n            plt.title(name)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.571264Z","iopub.status.idle":"2025-12-04T15:00:53.57152Z","shell.execute_reply.started":"2025-12-04T15:00:53.571382Z","shell.execute_reply":"2025-12-04T15:00:53.571391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_ds = train_ds.take(100)\nbatch = next(iter(sample_ds))\nplot_batch(batch, label2name=CFG.label2name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.572049Z","iopub.status.idle":"2025-12-04T15:00:53.572261Z","shell.execute_reply.started":"2025-12-04T15:00:53.572164Z","shell.execute_reply":"2025-12-04T15:00:53.572172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODELING","metadata":{}},{"cell_type":"code","source":"from keras.metrics import Precision, Recall, F1Score, TopKCategoricalAccuracy, AUC","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.57314Z","iopub.status.idle":"2025-12-04T15:00:53.573486Z","shell.execute_reply.started":"2025-12-04T15:00:53.573323Z","shell.execute_reply":"2025-12-04T15:00:53.573336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model():\n    \n    backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(CFG.preset)\n    backbone.trainable = True\n    \n    classifier = keras_cv.models.ImageClassifier(\n        backbone=backbone,\n        num_classes=CFG.num_classes,\n        name=\"classifier\",\n    )\n    \n    inp = keras.layers.Input(shape=(None, None, 3))\n    out = classifier(inp)\n    return keras.models.Model(inputs=inp, outputs=out)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.574583Z","iopub.status.idle":"2025-12-04T15:00:53.574794Z","shell.execute_reply.started":"2025-12-04T15:00:53.574697Z","shell.execute_reply":"2025-12-04T15:00:53.574706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = get_model()\nmodel.compile(\n    optimizer='adam',\n    loss=keras.losses.CategoricalCrossentropy(label_smoothing=0.02),\n    metrics=[\n        Precision(name='precision'),\n        Recall(name='recall'), \n        F1Score(name='f1_score', average='macro'),\n        TopKCategoricalAccuracy(name='top_k_cat_acc')\n    ],\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.575706Z","iopub.status.idle":"2025-12-04T15:00:53.575994Z","shell.execute_reply.started":"2025-12-04T15:00:53.575826Z","shell.execute_reply":"2025-12-04T15:00:53.575842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.576922Z","iopub.status.idle":"2025-12-04T15:00:53.57731Z","shell.execute_reply.started":"2025-12-04T15:00:53.577042Z","shell.execute_reply":"2025-12-04T15:00:53.577057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LR Schedule\n\nLearning Rate scheduler for transfer learning.\nThe learning rate starts from lr_start, then decreases to alr_min using different methods namely,\n* step: Reduce lr step wise like stair.\n* cos: Follow Cosine graph to reduce lr.\n* exp: Reduce lr exponentially.","metadata":{}},{"cell_type":"code","source":"import math\n\ndef get_lr_callback(batch_size=8, mode='cos', epochs=10, plot=False):\n    lr_start, lr_max, lr_min = 5e-5, 8e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        elif mode == 'exp': lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n        elif mode == 'step': lr = lr_max * lr_decay**((epoch - lr_ramp_ep - lr_sus_ep) // 2)\n        elif mode == 'cos':\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        return lr\n\n    if plot:  # Plot lr curve if plot is True\n        plt.figure(figsize=(10, 5))\n        plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('lr')\n        plt.title('LR Scheduler')\n        plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  # Create lr callback","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.578739Z","iopub.status.idle":"2025-12-04T15:00:53.578948Z","shell.execute_reply.started":"2025-12-04T15:00:53.578832Z","shell.execute_reply":"2025-12-04T15:00:53.57884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lr_cb = get_lr_callback(CFG.batch_size, plot=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.579321Z","iopub.status.idle":"2025-12-04T15:00:53.579577Z","shell.execute_reply.started":"2025-12-04T15:00:53.579455Z","shell.execute_reply":"2025-12-04T15:00:53.579468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Checkpoint","metadata":{}},{"cell_type":"code","source":"ckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model_15s.weights.h5\",\n                                         monitor='val_f1_score',\n                                         save_best_only=True,\n                                         save_weights_only=True,\n                                         mode='max')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.579944Z","iopub.status.idle":"2025-12-04T15:00:53.58021Z","shell.execute_reply.started":"2025-12-04T15:00:53.580063Z","shell.execute_reply":"2025-12-04T15:00:53.580074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    validation_data=test_ds, \n    epochs=CFG.epochs,\n    callbacks=[lr_cb, ckpt_cb], \n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.581157Z","iopub.status.idle":"2025-12-04T15:00:53.581354Z","shell.execute_reply.started":"2025-12-04T15:00:53.581257Z","shell.execute_reply":"2025-12-04T15:00:53.581265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Result Summary","metadata":{}},{"cell_type":"code","source":"best_epoch = np.argmax(history.history[\"val_f1_score\"])\nbest_score = history.history[\"val_f1_score\"][best_epoch]\nprint('>>> Best Epoch: ', best_epoch+1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.582178Z","iopub.status.idle":"2025-12-04T15:00:53.582403Z","shell.execute_reply.started":"2025-12-04T15:00:53.582303Z","shell.execute_reply":"2025-12-04T15:00:53.582312Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Graphics","metadata":{}},{"cell_type":"code","source":"# loss = history.history['loss']\n# val_loss = history.history['val_loss']\n\n# precision = history.history['precision']\n# val_precision = history.history['val_precision']\n\n# recall = history.history['recall']\n# val_recall = history.history['val_recall']\n\n# f1_score = history.history['f1_score']\n# val_f1_score = history.history['val_f1_score']\n\n# top_k_cat_acc = history.history['top_k_cat_acc']\n# val_top_k_cat_acc = history.history['val_top_k_cat_acc']\n\n# epochs_range = range(1, len(precision) + 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.583727Z","iopub.status.idle":"2025-12-04T15:00:53.584291Z","shell.execute_reply.started":"2025-12-04T15:00:53.584146Z","shell.execute_reply":"2025-12-04T15:00:53.584163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(12, 15))\n\n# plt.subplot(3, 2, 1)\n\n# plt.plot(epochs_range, loss, label='Entrenamiento')\n# plt.plot(epochs_range, val_loss, label='Validación')\n# plt.title('Loss')\n# plt.xlabel('Epoch')\n# plt.ylabel('Loss')\n# plt.legend()\n# plt.grid()\n\n# plt.subplot(3, 2, 2)\n# plt.plot(epochs_range, precision, label='Entrenamiento')\n# plt.plot(epochs_range, val_precision, label='Validación')\n# plt.title('Precision')\n# plt.xlabel('Epoch')\n# plt.ylabel('Presicion')\n# plt.legend()\n# plt.grid()\n\n\n# plt.subplot(3, 2, 3)\n# plt.plot(epochs_range, recall, label='Entrenamiento')\n# plt.plot(epochs_range, val_recall, label='Validación')\n# plt.title('Recall')\n# plt.xlabel('Epoch')\n# plt.ylabel('Recall')\n# plt.legend()\n# plt.grid()\n\n# plt.subplot(3, 2, 4)\n# plt.plot(epochs_range, f1_score, label='Entrenamiento')\n# plt.plot(epochs_range, val_f1_score, label='Validación')\n# plt.title('F1 Score')\n# plt.xlabel('Epoch')\n# plt.ylabel('F1 Score')\n# plt.legend()\n# plt.grid()\n\n\n# plt.subplot(3, 2, 5)\n# plt.plot(epochs_range, top_k_cat_acc, label='Entrenamiento')\n# plt.plot(epochs_range, val_top_k_cat_acc, label='Validación')\n# plt.title('Top K category Accuracy')\n# plt.xlabel('Epoch')\n# plt.ylabel('op K category Accuracy')\n# plt.legend()\n# plt.grid()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.584907Z","iopub.status.idle":"2025-12-04T15:00:53.585137Z","shell.execute_reply.started":"2025-12-04T15:00:53.585016Z","shell.execute_reply":"2025-12-04T15:00:53.585028Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TESTING","metadata":{}},{"cell_type":"markdown","source":"## Load Model\n\nExecute this cell to load model weights","metadata":{}},{"cell_type":"code","source":"#model = get_model()\n#model.load_weights(\"/kaggle/input/lz_birdclef2024_weights/keras/15s-model/1/best_model_15s.weights.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.586389Z","iopub.status.idle":"2025-12-04T15:00:53.586706Z","shell.execute_reply.started":"2025-12-04T15:00:53.586563Z","shell.execute_reply":"2025-12-04T15:00:53.586578Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Audio file in a DataFrame","metadata":{}},{"cell_type":"markdown","source":"## Test Data Pipeline","metadata":{}},{"cell_type":"code","source":"def load_ogg_test(path, target_sr=CFG.sample_rate, audio_len=CFG.audio_len):\n    def _load_ogg_np(filepath_tensor, sr_val):\n        filepath = filepath_tensor.numpy().decode('utf-8') \n        audio, _ = librosa.load(filepath, sr=sr_val, mono=True)\n        return audio.astype(np.float32)\n    \n    \n    audio_np = tf.py_function(func=_load_ogg_np, \n                              inp=[path, target_sr], \n                              Tout=tf.float32)\n    return audio_np\n\ndef create_frames(audio, duration=5, sr=CFG.sample_rate):\n    frame_size = int(duration * sr)\n    \n    audio_length = tf.shape(audio)[0]\n    remainder = audio_length % frame_size\n    pad_size = tf.cond(remainder > 0, lambda: frame_size - remainder, lambda: 0)\n    \n    audio = tf.pad(audio, [[0, pad_size]])  # Pad at the end to make divisible by frame_size\n    \n    frames = tf.reshape(audio, [-1, frame_size])  # shape: [num_frames, frame_size]\n    return frames\n\n\ndef preprocess_test(path):\n    audio = load_ogg_test(path, CFG.sample_rate)\n    audio_slices = create_frames(audio)\n\n \n    spec = keras.layers.MelSpectrogram( num_mel_bins=CFG.img_size[0],\n                                        fft_length=CFG.nfft, \n                                        sequence_stride=CFG.hop_length, \n                                        sampling_rate=CFG.sample_rate)(audio_slices)\n\n    spec = standarize_spec(spec)\n    \n    # Covnert spectrogram to 3 channel image (for imagenet)\n    spec = tf.tile(spec[..., None], [1, 1, 1, 3])\n    return spec\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.587632Z","iopub.status.idle":"2025-12-04T15:00:53.587903Z","shell.execute_reply.started":"2025-12-04T15:00:53.587761Z","shell.execute_reply":"2025-12-04T15:00:53.587774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_dataset_test(paths, batch_size=1):\n \n    slices = (paths, )\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    \n    # Preprocess each sample\n    ds = ds.map(preprocess_test, num_parallel_calls=tf.data.AUTOTUNE)\n    \n    ds = ds.batch(batch_size)\n\n    # Prefetch for performance\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    \n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.735578Z","iopub.execute_input":"2025-12-04T15:00:53.735872Z","iopub.status.idle":"2025-12-04T15:00:53.7407Z","shell.execute_reply.started":"2025-12-04T15:00:53.73585Z","shell.execute_reply":"2025-12-04T15:00:53.740015Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build DS","metadata":{}},{"cell_type":"code","source":"def get_testing_ds(path, samples=None):\n    \n    test_paths = glob(f'{path}/*ogg')\n    if samples:\n        test_paths = test_paths[:samples]\n    testing_df = pd.DataFrame(test_paths, columns=['filepath'])\n    \n    # Build test dataset\n    paths_list = testing_df.filepath.tolist()\n    ds = build_dataset_test(paths=paths_list, batch_size=1)\n\n    return  ds, paths_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.741671Z","iopub.execute_input":"2025-12-04T15:00:53.741903Z","iopub.status.idle":"2025-12-04T15:00:53.75581Z","shell.execute_reply.started":"2025-12-04T15:00:53.741883Z","shell.execute_reply":"2025-12-04T15:00:53.755181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ul_ds, paths_list = get_testing_ds(UNLABELED_AUDIO, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.756509Z","iopub.execute_input":"2025-12-04T15:00:53.756758Z","iopub.status.idle":"2025-12-04T15:00:53.77303Z","shell.execute_reply.started":"2025-12-04T15:00:53.756743Z","shell.execute_reply":"2025-12-04T15:00:53.771589Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Get Predictions","metadata":{}},{"cell_type":"code","source":"def get_predictions(ds, paths, n_audios):\n    # Initialize empty list to store ids\n    ids = []\n    \n    # Initialize empty array to store predictions\n    preds = np.empty(shape=(0, CFG.num_classes), dtype='float32')\n    \n    \n    # Iterate over each audio file in the test dataset\n    for idx, specs in enumerate(tqdm(iter(ds), desc='test ', total=n_audios)):\n        # Extract the filename without the extension\n        filename = paths[idx].split('/')[-1].replace('.ogg','')\n        \n        # Convert to backend-specific tensor while excluding extra dimension\n        specs = keras.ops.convert_to_tensor(specs[0])\n        \n        # Predict bird species for all frames in a recording using all trained models\n        frame_preds = model.predict(specs, verbose=0)\n        \n        # Create a ID for each frame in a recording using the filename and frame number\n        frame_ids = [f'{filename}_{(frame_id+1)*5}' for frame_id in range(len(frame_preds))]\n        \n        # Concatenate the ids\n        ids += frame_ids\n        # Concatenate the predictions\n        preds = np.concatenate([preds, frame_preds], axis=0)\n\n\n    return ids, preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.773497Z","iopub.status.idle":"2025-12-04T15:00:53.773781Z","shell.execute_reply.started":"2025-12-04T15:00:53.773635Z","shell.execute_reply":"2025-12-04T15:00:53.773648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_preds_df(ids, preds):\n    pred_df = pd.DataFrame(ids, columns=['row_id'])\n    pred_df.loc[:, CFG.class_names] = preds\n\n    pred_df_label = pd.DataFrame(ids, columns=['row_id'])\n    pred_df_label['max_pred_id'] = np.argmax(preds, axis=1)\n    pred_df_label['label'] = pred_df_label.max_pred_id.map(lambda x: CFG.label2name[x])\n    pred_df_label['common_name'] =  pred_df_label.label.map(lambda x: taxonomy.loc[x].PRIMARY_COM_NAME)\n    pred_df_label['scientific_name'] =  pred_df_label.label.map(lambda x: taxonomy.loc[x].SCI_NAME)  \n\n    return pred_df, pred_df_label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.775074Z","iopub.status.idle":"2025-12-04T15:00:53.775298Z","shell.execute_reply.started":"2025-12-04T15:00:53.775198Z","shell.execute_reply":"2025-12-04T15:00:53.775206Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Predictions Result","metadata":{}},{"cell_type":"code","source":"ul_ids, ul_preds = get_predictions(test_ul_ds, paths_list, len(paths_list))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.776404Z","iopub.status.idle":"2025-12-04T15:00:53.776686Z","shell.execute_reply.started":"2025-12-04T15:00:53.776573Z","shell.execute_reply":"2025-12-04T15:00:53.776586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ipd.Audio(paths_list[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.778413Z","iopub.status.idle":"2025-12-04T15:00:53.778901Z","shell.execute_reply.started":"2025-12-04T15:00:53.778735Z","shell.execute_reply":"2025-12-04T15:00:53.778749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_df, pred_df_label = get_preds_df(ul_ids, ul_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.779351Z","iopub.status.idle":"2025-12-04T15:00:53.779611Z","shell.execute_reply.started":"2025-12-04T15:00:53.779467Z","shell.execute_reply":"2025-12-04T15:00:53.779481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.781389Z","iopub.status.idle":"2025-12-04T15:00:53.781694Z","shell.execute_reply.started":"2025-12-04T15:00:53.781554Z","shell.execute_reply":"2025-12-04T15:00:53.781571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_df_label.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.782699Z","iopub.status.idle":"2025-12-04T15:00:53.782896Z","shell.execute_reply.started":"2025-12-04T15:00:53.7828Z","shell.execute_reply":"2025-12-04T15:00:53.782809Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Individiual Testing","metadata":{}},{"cell_type":"code","source":"TESTING = '/kaggle/input/birds-birdclef2024-test'\ntest_ds, test_pl = get_testing_ds(TESTING)\ntest_ids, test_preds = get_predictions(test_ds, test_pl, len(test_pl))\ntest_pred_df, test_pred_df_label = get_preds_df(test_ids, test_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.783885Z","iopub.status.idle":"2025-12-04T15:00:53.784169Z","shell.execute_reply.started":"2025-12-04T15:00:53.784025Z","shell.execute_reply":"2025-12-04T15:00:53.784037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.78474Z","iopub.status.idle":"2025-12-04T15:00:53.78502Z","shell.execute_reply.started":"2025-12-04T15:00:53.784877Z","shell.execute_reply":"2025-12-04T15:00:53.78489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" test_pred_df_label.head(50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-04T15:00:53.786124Z","iopub.status.idle":"2025-12-04T15:00:53.786324Z","shell.execute_reply.started":"2025-12-04T15:00:53.786225Z","shell.execute_reply":"2025-12-04T15:00:53.786234Z"}},"outputs":[],"execution_count":null}]}