{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11773361,"sourceType":"datasetVersion","datasetId":7244978}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# %config Completer.use_jedi = False","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Todo:\n- Metrics for multi-label classification problem\n- Test with the 5 second chunks of the train audio.\n- We can test with the train soundscape as well to know any other data format we need to do before submission\n\n\nSample notebook submissionfrom the test_soundscaptes folder: https://www.kaggle.com/code/stefankahl/birdclef-2025-sample-submission\n","metadata":{}},{"cell_type":"code","source":"# # Reading in meta-data file\n\n# import numpy as np\n# import pandas as pd\n# meta_data = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# meta_data.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# meta_data.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import librosa\n# import numpy as np\n# import pandas as pd","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Load in the soundscape data with librosa understand the data we're dealing with\n# # What does librosa.load really return: \n# # https://stackoverflow.com/questions/61986490/what-does-librosa-load-return\n# np.random.seed(42)\n\n# signal, rate = librosa.load(path=\"/kaggle/input/birdclef-2025/train_audio/1139490/CSA36385.ogg\", sr=None)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from IPython.display import Audio\n\n# Audio(data=signal, rate=rate)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# sampling_rate = rate\n# time_step = 1 / sampling_rate\n\n# #Generate time vector\n# time = np.linspace(0, len(signal)/rate, num=len(signal))\n\n# plt.figure(figsize=(10, 4))\n# plt.plot(time, signal)\n# plt.xlabel(\"Time (seconds)\")\n# plt.ylabel(\"Amplitude\")\n# plt.title(\"Audio wavefrom after transformation\")\n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(12, 4))\n# librosa.display.waveshow(signal, sr=sampling_rate, axis='time')\n# plt.title('Audio Waveform')\n# plt.xlabel('Time (seconds)')\n# plt.ylabel('Amplitude')\n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FRAME_SIZE = 2048\n# HOP_SIZE = 512\n# stft_data = librosa.stft(signal, n_fft=FRAME_SIZE, hop_length=HOP_SIZE)\n# stft_data.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# stft_data_ammplitude = np.abs(stft_data) ** 2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Visualizating the spectrogram\n# plt.figure(figsize=(25, 10))\n# librosa.display.specshow(stft_data_ammplitude, \n#                          sr=sampling_rate,\n#                          hop_length=HOP_SIZE,\n#                          x_axis=\"time\", \n#                          y_axis=\"linear\")\n# plt.colorbar(format=\"%+2.f\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Let's do it in log scale\n# plt.figure(figsize=(25, 10))\n# librosa.display.specshow(librosa.power_to_db(stft_data_ammplitude), \n#                          sr=sampling_rate,\n#                          hop_length=HOP_SIZE,\n#                          x_axis=\"time\", \n#                          y_axis=\"linear\")\n# plt.colorbar(format=\"%+2.f\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plt.figure(figsize=(25, 10))\n# librosa.display.specshow(librosa.power_to_db(stft_data_ammplitude), \n#                          sr=sampling_rate,\n#                          hop_length=HOP_SIZE,\n#                          x_axis=\"time\", \n#                          y_axis=\"log\")\n# plt.colorbar(format=\"%+2.f\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# mffc_features = librosa.feature.mfcc(y=signal, sr=sampling_rate)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# with open(\"/kaggle/working/audio_file_name.txt\", \"w\") as a_file:\n#     for path, subdirs, files in os.walk('/kaggle/input/birdclef-2025/train_audio/'):\n#         for filename in files:\n#             f = os.path.join(path, filename)\n#             a_file.write(str(f) + os.linesep)\n            ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\ndirs = []\nfor directory in os.listdir('/kaggle/input/birdclef-2025/train_audio/'):\n    dirs.append(directory)\ndirs.sort()\nprint(dirs)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport math, random\nimport torch\nimport torchaudio\nfrom torchaudio import transforms\nimport matplotlib.pyplot as plt\nimport ast\nfrom IPython.display import Audio, display\nfrom torch.utils.data import DataLoader, Dataset, random_split\n\nimport torch.nn.functional as F\nimport torch.nn as nn\nfrom torch.nn import init","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Input for the filepath\n# train_folder = \"/kaggle/input/birdclef-2025/train_audio\"\n# label = \"creoro1\"\n# i = 0\n# for file in sorted(os.listdir(os.path.join(train_folder, label))):\n#     signal, rate = librosa.load(os.path.join(train_folder, label, file), sr=None)\n#     print(f\" The is file {file} with sampling rate of {rate}\")\n#     display(Audio(data=signal, rate=rate))\n#     i += 1\n#     if i == 15:\n#         break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"End-to-end model building\n\nhttps://towardsdatascience.com/audio-deep-learning-made-simple-sound-classification-step-by-step-cebc936bbe5/","metadata":{}},{"cell_type":"code","source":"train_label_1 = pd.read_csv(\"/kaggle/input/birdclef-label/train_label_1.csv\")\ntrain_label_2 = pd.read_csv(\"/kaggle/input/birdclef-label/train_label_2.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_1.dropna(subset=[\"offset_time\"], inplace=True)\ntrain_label_2.dropna(subset=[\"offset_time\"], inplace=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(len(set(train_label_1[\"primary_label\"])))\n# print(len(set(train_label_2[\"primary_label\"])))\nprint(sorted(list(set(train_label_1[\"primary_label\"]))))\nprint(sorted(list(set(train_label_2[\"primary_label\"]))))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label = pd.concat([train_label_1, train_label_2], axis=0)\ntrain_label.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo1: Build the meta data csv file\ncolumns_name = [\"FileName\", \"Offset\", \"Labels\"]\ntrain_df = pd.DataFrame(columns=columns_name)\n\nfor index, row in train_label.iterrows():\n    offset_time = str(row[\"offset_time\"]).split(\",\")\n    labels = tuple([str(row['primary_label'])] + ast.literal_eval(row['secondary_labels']))\n    \n    for offset in offset_time:\n        new_row = pd.DataFrame([[row['filename'], offset, labels]], columns=columns_name)\n        train_df = pd.concat([train_df, new_row], axis=0)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Inivestigate the issues with how I lable the data and fix it\n# list(train_df[\"Offset\"].unique())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Couple of issues:\n# 1) drop the missing offset data\n# 2) one instance where the offset time is separate by dot instead of comma, separate that\n# 3) convert everything in offset column from string type to integer type\n# 4) Reset the index, very important later when the Deep Learning model do the fit\n\n# 1\ntrain_df = train_df[train_df['Offset'] != ' ']\n\n# 2\nweird_instance = train_df[train_df[\"Offset\"] == \"5. 25\"].reset_index(drop=True)\nfor offset in [5, 25]:\n    new_row = pd.DataFrame([[weird_instance.loc[0,\"FileName\"], offset, weird_instance.loc[0,\"Labels\"]]], columns=columns_name)\n    train_df = pd.concat([train_df, new_row], axis=0)\ntrain_df = train_df[train_df['Offset'] != '5. 25']\n\n# 3\ntrain_df[\"Offset\"] = train_df[\"Offset\"].astype(int)\n\n# 4\ntrain_df.reset_index(drop=True, inplace=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nmlb = MultiLabelBinarizer(classes=dirs)\nmlb.fit(train_df[\"Labels\"])\ntrain_df_labels = mlb.transform(train_df[\"Labels\"])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo2: Transform data before feeding into the model. Those includes\n# 1) Load the files\n# 2) Resample and convert to stereo. I know all the data I'm fitting in is 32kHz sampling rate, \n#    but I need to double check whether it's mono (1 channel) or stereo (2 channel). You can\n#    know it by the shape of the signal\n# 3) Resize to fixed length. I already know that I only want a couple of 5 second long portion of each file\n# 4) Data Augmentation to create more data. Do I need this\n# 5) Feature Extraction: Ok now the data is clean and the latest techniques is to convert it to mel spectrogram\n# 6) Another round of data augmentation with masking\nclass AudioUtil():\n    @staticmethod\n    def load_audio(path, offset=None, duration=None):\n        # Basically load with native sampling rate\n        signal, sampling_rate = librosa.load(path, offset=offset, duration=duration, sr=None)\n        return (torch.from_numpy(signal.reshape(1, -1)), sampling_rate)\n\n    @staticmethod\n    def rechannel(audio, new_channel):\n        '''\n        audio: \n            - signal (either mono or stereo)\n            - sampling_rate - in Hrtz\n        new_channel: number of new channel we want to conver to\n        '''\n        sig, sr = audio\n        if sig.shape[0] == new_channel:\n            return audio\n\n\n        if new_channel == 1:\n            # Convert to mono if only usinng the first channel\n            resig = sig[:1, :]\n        else:\n            # Duplicate the first channel to make it stereo\n            resig = torch.cat((sig, sig))\n        return resig\n\n    @staticmethod\n    def resample(audio, newsr):\n        '''\n        audio: \n            - signal (either mono or stereo)\n            - sampling_rate - in Hrtz\n        newsr: number of new channel we want to convert to\n        '''\n        sig, sr = audio\n        if sr == newsr:\n            return audio\n\n        num_channels = sig.shape[0]\n        resig = torchaudio.transform.Resample(sr, newsr)(sig[:1, :])\n        if(num_channels > 1):\n            retwo = torchaudio.transform.Resample(sr, newsr)(sig[:1, :])\n            resig = torch.cat([resig, retwo])\n\n        return ((resig, newsr))\n\n    @staticmethod\n    def pad_trunc(aud, max_ms):\n        sig, sr = aud\n        num_rows, sig_len = sig.shape\n        max_len = sr//1000 * max_ms\n\n        if (sig_len >= max_len):\n            # truncat\n            sig = sig[:,:max_len]\n        else:\n            # pad\n            pad_begin_len = random.randint(0, max_len - sig_len)\n            pad_end_len = max_len - sig_len - pad_begin_len\n            pad_begin = torch.zeros((num_rows, pad_begin_len))\n            pad_end = torch.zeros((num_rows, pad_end_len))\n    \n            sig = torch.cat((pad_begin, sig, pad_end), 1)\n\n        return (sig, sr)\n\n    @staticmethod\n    def time_shift(aud, shift_limit):\n        sig, sr = aud\n        _, sig_len = sig.shape\n        shift_amt = int(random.random() * shift_limit * sig_len)\n        return (sig.roll(shift_amt), sr)\n\n    @staticmethod\n    def spectro_gram(aud, n_mels=64, n_fft=2048, hop_len=None):\n        sig, sr = aud\n        top_db = 80\n\n        # spectrogram shape [channel, n_mels, time] where channel is mono, stereo\n        spec = transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n        # convert to decibels\n        spec = transforms.AmplitudeToDB(top_db=top_db)(spec)\n        return (spec)\n\n    \n    def spectro_augment(spec, max_mask_pct=0.1, n_freq_mask=1, n_time_mask=1):\n        _, n_mels, n_steps = spec.shape\n        mask_value = spec.mean()\n        aug_spec = spec\n            \n        freq_mask_param = max_mask_pct * n_mels\n        for _ in range(n_freq_masks):\n          aug_spec = transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n    \n        time_mask_param = max_mask_pct * n_steps\n        for _ in range(n_time_masks):\n          aug_spec = transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n    \n        return aug_spec\n\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo3: Let's build the Dataset. One of the building block for Pytorch \n# We want this because we can't just load everything in RAM. For deep learning, just load it while training\nclass SoundDS(Dataset):\n    def __init__(self, df, df_labels, data_path):\n        self.df = df\n        self.data_path = str(data_path)\n        self.duration = 5000 # 5 second long\n        self.sr = 32000 # sampling rate provided in 32 kHz\n        self.channel = 1\n        self.shift_pct = 0.4\n        self.df_labels = df_labels\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        '''\n        Purpose of this function is to get the data of the file and corresponding ID\n        '''\n        audio_file = os.path.join(self.data_path, self.df.loc[idx, 'FileName'])\n        class_id = torch.tensor(self.df_labels[idx, :], dtype=torch.float)\n\n        aud = AudioUtil.load_audio(audio_file, offset=self.df.loc[idx, 'Offset'], duration=5)\n        reaud = AudioUtil.resample(aud, self.sr)\n        rechan= AudioUtil.rechannel(reaud, self.channel)\n        dur_aud = AudioUtil.pad_trunc(rechan, self.duration)\n        sgram = AudioUtil.spectro_gram(dur_aud, n_mels=128, n_fft=1024, hop_len=None)\n        \n        return sgram, class_id\n      ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo 4: Prepare batches of data\ntrain_folder = \"/kaggle/input/birdclef-2025/train_audio\"\nmyds = SoundDS(train_df, train_df_labels, train_folder)\n\n# Random split of 80:20\nnum_items = len(myds)\nnum_train = round(num_items * 0.8)\nnum_val = num_items - num_train\ntrain_ds, val_ds = random_split(myds, [num_train, num_val], generator=torch.Generator().manual_seed(42))\n\n# Training and validation data loaders\ntrain_dl = DataLoader(train_ds, batch_size=16, shuffle=True)\nval_dl = DataLoader(val_ds, batch_size=16, shuffle=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo 5: Create Model, CNN with 4 convoluted layer\nclass AudioClassifer(nn.Module):\n    def __init__(self):\n        super().__init__()\n        conv_layers= []\n\n        # First convolution block with Relu and Batch Norm. Use Kaiming initialization\n        self.conv1 = nn.Conv2d(1, 8, kernel_size=(5,5), stride=(2,2), padding=(2,2))\n        self.relu1 = nn.ReLU()\n        self.bn1 = nn.BatchNorm2d(8)\n        init.kaiming_normal_(self.conv1.weight, a=0.1)\n        self.conv1.bias.data.zero_()\n        conv_layers += [self.conv1, self.relu1, self.bn1]\n\n        # Second Convolution Block\n        self.conv2 = nn.Conv2d(8, 16, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu2 = nn.ReLU()\n        self.bn2 = nn.BatchNorm2d(16)\n        init.kaiming_normal_(self.conv2.weight, a=0.1)\n        self.conv2.bias.data.zero_()\n        conv_layers += [self.conv2, self.relu2, self.bn2]\n\n        # Third Convolution Block\n        self.conv3 = nn.Conv2d(16, 32, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu3 = nn.ReLU()\n        self.bn3 = nn.BatchNorm2d(32)\n        init.kaiming_normal_(self.conv3.weight, a=0.1)\n        self.conv3.bias.data.zero_()\n        conv_layers += [self.conv3, self.relu3, self.bn3]\n\n        # Fourth Convolution Block\n        self.conv4 = nn.Conv2d(32, 64, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu4 = nn.ReLU()\n        self.bn4 = nn.BatchNorm2d(64)\n        init.kaiming_normal_(self.conv4.weight, a=0.1)\n        self.conv4.bias.data.zero_()\n        conv_layers += [self.conv4, self.relu4, self.bn4]\n\n        # Linear Classifier\n        self.ap = nn.AdaptiveAvgPool2d(output_size=1)\n        self.lin = nn.Linear(in_features=64, out_features=206)\n        self.sigmoid = nn.Sigmoid()\n\n        # Wrap everything together\n        self.conv = nn.Sequential(*conv_layers)\n\n    def forward(self, x):\n        # run convolution blocks\n        x = self.conv(x)\n\n        # adaptive pooling and flatten for input to linear layer\n        x = self.ap(x)\n        x = x.view(x.shape[0], -1)\n\n        # Linear Layer and then sigomoid?\n        x = self.lin(x)\n        x = self.sigmoid(x)\n        return x\n\nmyModel = AudioClassifer()\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmyModel = myModel.to(device)\n\n# Check that it is on Cuda\nnext(myModel.parameters()).device","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo 6: Define the evaluation metrics. Macro-avearged ROC-AUC score\nfrom sklearn.metrics import roc_auc_score\ndef macro_roc_auc_score(y_true, y_pred_proba):\n    n_classes = y_true.shape[1]\n    auc_scores = []\n    for i in range(n_classes):\n        if np.unique(y_true[:,i]).size > 1:\n            auc = roc_auc_score(y_true[:,i], y_pred_proba[:,i])\n            auc_scores.append(auc)\n    return np.mean(auc_scores)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo 6: Training\ndef training(model, train_dl, num_epochs):\n    criterion = nn.BCELoss()\n\n    optimizer = torch.optim.Adam(model.parameters(),lr=0.007)\n    scheduler = torch.optim.lr_scheduler.OneCycleLR(optimizer, \n                                                    max_lr=0.007,\n                                                    steps_per_epoch=int(len(train_dl)),\n                                                    epochs=num_epochs,\n                                                    anneal_strategy='cos')\n\n    for epoch in range(num_epochs):\n        running_loss = 0.0\n        correct_prediction = 0\n\n        # Repeat for each batch in training set\n        for i, data in enumerate(train_dl):\n            inputs, labels = data[0].to(device), data[1].to(device)\n\n            # Normalize data\n            inputs_m, inputs_s = inputs.mean(), inputs.std()\n            inputs = (inputs - inputs_m) / inputs_s\n        \n            # Zero parameter gradients\n            optimizer.zero_grad()\n\n            # forward + backward + optimize\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            scheduler.step()\n            \n            # Keep stats for Loss and Accuracy\n            running_loss += loss.item()\n        \n        avg_loss = running_loss / len(train_dl)\n        print(f'Epoch: {epoch}, Loss: {avg_loss:.5f}')\n    print(\"Finished Training\")\n\nstart_time = time.time()\nnum_epochs = 110\ntraining(myModel, train_dl, num_epochs)\nend_time = time.time()\nprint(f\"Elapsed time for training: {end_time - start_time} seconds\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Todo 7: Inference for one instance. Helpful for submitting the model\n# val_features, val_labels = next(iter(val_dl))\n# with torch.no_grad():\n#     inputs, labels = val_features[0].to(device), val_labels[0].to(device)\n#     inputs = inputs.unsqueeze(0)\n    \n#     # Normalize the inputs\n#     inputs_m, inputs_s = inputs.mean(), inputs.std()\n#     inputs = (inputs - inputs_m) / inputs_s\n\n#     outputs = myModel(inputs)\n#     print(f\"Shape of outputs: {outputs.shape}\")\n#     print(f\"Shape of labels: {labels.shape}\")\n    \n#     # print(torch.flatten(outputs).tolist())\n    \n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo 7: Inference for the whole val_ds DataLoader instance. Helpful for submitting the model\nfrom torchmetrics.classification import MultilabelAUROC\ncriterion = nn.BCELoss()\nval_loss = 0.0\nall_outputs = []\nall_labels = []\nwith torch.no_grad():\n    for data in val_dl:\n        inputs, labels = data[0].to(device), data[1].to(device)\n\n        # Normalize the inputs\n        inputs_m, inputs_s = inputs.mean(), inputs.std()\n        inputs = (inputs - inputs_m) / inputs_s\n    \n        outputs = myModel(inputs)\n        loss = criterion(outputs, labels)\n        val_loss += loss.item()\n        # print(f\"Shape of outputs: {outputs.shape}\")\n        # print(f\"Shape of labels: {labels.shape}\")\n        all_outputs.append(outputs.cpu())\n        all_labels.append(labels.cpu())\nall_outputs = torch.cat(all_outputs, dim=0)\nall_labels = torch.cat(all_labels, dim=0)\n\nnum_labels = all_outputs.shape[1]\nauroc_metric = MultilabelAUROC(num_labels=num_labels, average='macro')\n\nmacro_auc = auroc_metric(all_outputs, all_labels.int()).item()\n\nprint(f\"Avearge macro scores among all validation batches is: {macro_auc}\")\nprint(f\"Validation Loss is: {val_loss/len(val_dl)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo: Integrate validation loss\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Save the model\n# model_path = \"/kaggle/working/audio_classifier.pth\"\n# torch.save(myModel.state_dict(), model_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Load the model\n# model_path = \"/kaggle/input/bird_classification/pytorch/default/1/audio_classifier.pth\"\n# myModel = AudioClassifer()\n# myModel.load_state_dict(torch.load(model_path, weights_only=True))\n# myModel.eval()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Todo: Submit the prediction based upon the sample of the competition\n# Set seed\nnp.random.seed(42)\n\n# Class labels from train audio\nclass_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n\n# List of test soundscapes (only visible during submission)\ntest_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntest_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n\n# Open each soundscape and make predictions for 5-second segments\n# Use pandas df with 'row_id' plus class labels as columns\npredictions = pd.DataFrame(columns=['row_id'] + class_labels)\nfor soundscape in test_soundscapes:\n\n    # Load audio\n    sig, rate = librosa.load(path=soundscape, sr=None)\n\n    # Split into 5-second chunks\n    chunks = []\n    for i in range(0, len(sig), rate*5):\n        chunk = sig[i:i+rate*5]\n        chunks.append(chunk)\n        \n    # Make predictions for each chunk\n    for i, chunk in enumerate(chunks):\n        \n        # Get row id  (soundscape id + end time of 5s chunk)      \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n        \n        # Make prediction (let's use random scores for now)\n        # Preprocess the data > Feed the data into model > append prediction\n        chunk = torch.from_numpy(chunk.reshape(1, -1))\n        aud = (chunk, rate)\n        reaud = AudioUtil.resample(aud, rate)\n        rechan= AudioUtil.rechannel(reaud, 1)\n        dur_aud = AudioUtil.pad_trunc(rechan, 5000)\n        sgram = AudioUtil.spectro_gram(dur_aud, n_mels=64, n_fft=1024, hop_len=None)\n\n        with torch.no_grad():\n            inputs = sgram.unsqueeze(0)\n            \n            # Normalize the inputs\n            inputs_m, inputs_s = inputs.mean(), inputs.std()\n            inputs = (inputs - inputs_m) / inputs_s\n        \n            outputs = myModel(inputs)\n\n        \n        # Append to predictions as new row\n        new_row = pd.DataFrame([[row_id] + torch.flatten(outputs).tolist()], columns=['row_id'] + class_labels)\n        predictions = pd.concat([predictions, new_row], axis=0, ignore_index=True)\n        \n# Save prediction as csv\npredictions.to_csv('submission.csv', index=False)\npredictions.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}