{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":372688,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":308389,"modelId":328812},{"sourceId":373585,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":308961,"modelId":329363}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import the required libraries\n**Justification**\n\n* Os is used to walk through the audio folders to get file paths and labels.\n* Random is used to randomly select audio chunks to introduce variation during training and reduce overfitting.\n* Numpy is used to pad short audio chunks and normalize the spectrograms.\n* Librosa is used to load the .ogg files, compute the mel spectrograms and convert the scale into a decibel scale.\n* Torch is used to construct the CNN, defines the optimizer for the training prodecure, splits the data into train and validation set and is used to load the data.\n* Tqdm adds a progress bar to see the progress of training the CNN.\n* sklearns train test split is used to split the data into train and validation set\n* matplotlib is optionally used to create images for the poster\n* sklearn metrics is used to compute the classification metrics on the 20% validation set","metadata":{}},{"cell_type":"code","source":"import os \nimport random\nimport numpy as np\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, random_split\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T12:23:53.709218Z","iopub.execute_input":"2025-05-07T12:23:53.709658Z","iopub.status.idle":"2025-05-07T12:23:58.718223Z","shell.execute_reply.started":"2025-05-07T12:23:53.709617Z","shell.execute_reply":"2025-05-07T12:23:58.717264Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define parameters and helper functions \n* This part of code sets initial parameters that are later called in other parts of the script.\n* Three helper functions are defined of which the first splits the audios into 5 second chunks\n* The second function converts the audio parts into mel spectrograms\n* The third function normalizes the spectrograms","metadata":{}},{"cell_type":"code","source":"#Parameters\nAUDIO_DIR = '/kaggle/input/birdclef-2025/train_audio/' #locate the audio files\nSR        = 32000 #Sampling rate 32K hz\nCHUNK_LEN = 5  #5 seconds\nN_MELS    = 128 #the amount of mel frequency bands used to convert audios to spectrogram\nBATCH_SIZE = 32 #Choose batch size\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu') #Define when to use gpu or cpu\n\n###Helper functions\n#Split audio in 5 seconds function\ndef split_audio(audio, sr=SR, chunk_length=CHUNK_LEN):\n    samples_per_chunk = chunk_length * sr\n    chunks = []\n    for i in range(0, len(audio), samples_per_chunk):\n        chunk = audio[i:i + samples_per_chunk]\n        if len(chunk) < samples_per_chunk:\n            chunk = np.pad(chunk,\n                           (0, samples_per_chunk - len(chunk)),\n                           mode='constant')\n        chunks.append(chunk)\n    return chunks\n\n#Converts audio chunks into mel spectrograms\ndef to_mel_spectrogram(audio_chunk, sr=SR, n_mels=N_MELS):\n    mel = librosa.feature.melspectrogram(y=audio_chunk,\n                                         sr=sr,\n                                         n_mels=n_mels)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n    return mel_db\n\n#Normalizes all spectrograms created\ndef normalize_mel(mel_db):\n    mel_db -= mel_db.min()\n    max_val = mel_db.max()\n    if max_val > 0:\n        mel_db /= max_val\n    else:\n        mel_db[:] = 0.0\n    return mel_db","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T13:32:25.266674Z","iopub.execute_input":"2025-05-07T13:32:25.266986Z","iopub.status.idle":"2025-05-07T13:32:25.276872Z","shell.execute_reply.started":"2025-05-07T13:32:25.266964Z","shell.execute_reply":"2025-05-07T13:32:25.275279Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pre-process data\n* The code below defines how individual audio files are processes and loaded during the training process.\n* If the audios file length does not exactly round to the desired 5 seconds at the end the missing seconds will be padded with 0s.\n* Random 5 second chunks will be selected during the training process to make the model generalize better during the training process.","metadata":{}},{"cell_type":"code","source":"#Stores filepaths and labels inside BirdclefDataset object\nclass BirdclefDataset(Dataset):\n    def __init__(self, filepaths, labels):\n        self.filepaths = filepaths\n        self.labels = labels\n        \n#Returns the length of the amount of samples\n    def __len__(self):\n        return len(self.filepaths)\n\n##Audio processing\n    def __getitem__(self, idx):\n        filepath = self.filepaths[idx]\n        label = self.labels[idx]\n\n#Loads audio and splits with helper function\n        audio, _ = librosa.load(filepath, sr=SR)\n        chunks = split_audio(audio, sr=SR)\n\n        if len(chunks) == 0:\n            #If the audio file is shorter than CHUNK_LEN, pad it with 0s\n            chunk = np.pad(audio, (0, SR*CHUNK_LEN - len(audio)), mode='constant')\n        else:\n            chunk = random.choice(chunks)\n        #Convert chunks to actual spectrograms\n        mel = to_mel_spectrogram(chunk)\n        mel = normalize_mel(mel)\n        mel = torch.tensor(mel, dtype=torch.float32).unsqueeze(0)\n\n        return mel, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T12:24:04.401674Z","iopub.execute_input":"2025-05-07T12:24:04.401968Z","iopub.status.idle":"2025-05-07T12:24:04.408978Z","shell.execute_reply.started":"2025-05-07T12:24:04.401944Z","shell.execute_reply":"2025-05-07T12:24:04.408177Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CNN structure definition\n* The code below defines the custom convolutional neural network (CNN) which is used to train the data with. \n* The CNN consists of 3 convolutional layers and one linear layer. \n* For each layer the rectified linear unit activation function is used. In the first layer, from each spectrogram 16 features are extracted which are used as input in the second layer which extracts 32 features which is used for input in the third layer which outputs 64 features.","metadata":{}},{"cell_type":"code","source":"#CNN model layers\nclass CNNModel(nn.Module):\n    def __init__(self, num_classes):\n        super().__init__()\n        self.cnn = nn.Sequential(\n            nn.Conv2d(1, 16, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n\n            nn.Conv2d(16, 32, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n\n            nn.Conv2d(32, 64, 3, padding=1),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool2d((1, 1)),\n        )\n        self.fc = nn.Linear(64, num_classes)\n\n    #Define how input moves through layers\n    def forward(self, x):\n        x = self.cnn(x)\n        x = x.view(x.size(0), -1)\n        x = self.fc(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T12:24:08.010334Z","iopub.execute_input":"2025-05-07T12:24:08.010637Z","iopub.status.idle":"2025-05-07T12:24:08.018273Z","shell.execute_reply.started":"2025-05-07T12:24:08.010615Z","shell.execute_reply":"2025-05-07T12:24:08.016824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare dataset\n* This part of code saves the full audio paths and associated labels\n* The data is split into 80% training and 20% validation set using stratification","metadata":{}},{"cell_type":"code","source":"#Create 2 empty lists\nogg_paths = []\nlabels = []\n\n#This part walks through all audio files in the directory and extracts the full paths and associated labels\nfor root, _, files in os.walk(AUDIO_DIR):\n    for fname in files:\n        if fname.lower().endswith('.ogg'):\n            ogg_paths.append(os.path.join(root, fname))\n            label = os.path.basename(root)  #foldername = bird species\n            labels.append(label)\n\n#Map the bird labels to integers\nunique_labels = sorted(list(set(labels)))\nlabel2idx = {label: idx for idx, label in enumerate(unique_labels)}\nidx2label = {idx: label for label, idx in label2idx.items()}\nlabels_idx = [label2idx[label] for label in labels]\n\n#Print label order (later used for the actual submission script)\nprint(\"✅ unique_labels (label → index mapping):\")\nfor i, label in enumerate(unique_labels):\n    print(f\"{i}: {label}\")\n\nfull_dataset = BirdclefDataset(ogg_paths, labels_idx)\n\n#80% training and 20% validation set splitting\ntrain_paths, val_paths, train_labels, val_labels = train_test_split(\n    ogg_paths,\n    labels_idx,\n    test_size=0.2,\n    stratify=labels_idx,\n    random_state=42\n)\n\n#Create the datasets from created split\ntrain_dataset = BirdclefDataset(train_paths, train_labels)\nval_dataset   = BirdclefDataset(val_paths, val_labels)\n\n###This is the code for non-stratified splitting\n#train_size = int(0.8 * len(full_dataset))\n#val_size   = len(full_dataset) - train_size\n#train_dataset, val_dataset = random_split(full_dataset, [train_size, val_size])\n\n#These data loaders below will be used later during training to feed batches into the model and to evaluate the performance during training\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=0)\nval_loader   = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T12:24:10.822277Z","iopub.execute_input":"2025-05-07T12:24:10.822731Z","iopub.status.idle":"2025-05-07T12:25:02.620187Z","shell.execute_reply.started":"2025-05-07T12:24:10.822701Z","shell.execute_reply":"2025-05-07T12:25:02.619322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train model and save output (Not nessecary to run if you want to predict with the model thats already loaded in the environment)\n* The final part of the script performs the actual training of the model.\n* After each epoch a round of validation is conducted, it is important to note that this does not affect the weights of the model to prevent overfitting on the 20% validation set.\n* Finally, the model is saved","metadata":{}},{"cell_type":"code","source":"#Training the model\nmodel = CNNModel(num_classes=len(unique_labels)).to(device) #Initiliazes CNN model to output num_classes\noptimizer = optim.Adam(model.parameters(), lr=1e-3) #Adam optimizer is used for updating the model weights\ncriterion = nn.CrossEntropyLoss() #Loss function used for classification (cross entropy)\n\nfor epoch in range(10):  #Epoch number = 10, with 10 epochs this takes like 4 hours\n    model.train() #Sets model in training mode\n    train_loss = 0.0\n\n    for X_batch, y_batch in tqdm(train_loader, desc=f\"Epoch {epoch+1} [Train]\"):\n        X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(X_batch)\n        loss = criterion(outputs, y_batch)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item() * X_batch.size(0)\n\n    model.eval() #Sets the model in evaluation mode\n    val_loss = 0.0\n\n    with torch.no_grad(): #this makes sure the model validation is done without updating the weights\n        for X_batch, y_batch in tqdm(val_loader, desc=f\"Epoch {epoch+1} [Val]\"):\n            X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n            outputs = model(X_batch)\n            loss = criterion(outputs, y_batch)\n            val_loss += loss.item() * X_batch.size(0)\n\n    print(f\"Epoch {epoch+1}: Train loss {train_loss/len(train_dataset):.4f} | Val loss {val_loss/len(val_dataset):.4f}\") #Reports average training and validation loss for each epoch\n\n#Save the model\ntorch.save(model.state_dict(), '/kaggle/working/birdclef_cnn.pth')\nprint(\"✅ Model saved to /kaggle/working/birdclef_cnn.pth\") #To make sure the model is saved","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T18:13:56.098523Z","iopub.execute_input":"2025-04-28T18:13:56.099104Z","iopub.status.idle":"2025-04-28T22:16:40.661233Z","shell.execute_reply.started":"2025-04-28T18:13:56.099062Z","shell.execute_reply":"2025-04-28T22:16:40.660439Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Finally prediction on the 20% validation set\n* This part of the script will evaluate the prediction accuracy on all 5s audio chunks of the 20% validation dataset\n* First, the CNN model architecture is defined for the second time in case this is not already done by running the previous code blocks\n* SAVE_PATH is the location of the trained model. If there is no model trained, MODEL_PATH will be used for the prediction.\n* The MODEL_PATH model is the same model that you will obtain with running the whole script but it unfortunately takes about 4 hours to train\n* Luckily, I could uploaded the model in the kaggle environment and therefore it is available to use without having to wait these 4 hours\n* Additionally, a plot is created showing the class distribution between the training dataset and validation dataset","metadata":{}},{"cell_type":"code","source":"#Define the CNN model architecture if not already defined\ntry:\n    CNNModel\nexcept NameError:\n    class CNNModel(nn.Module):\n        def __init__(self, num_classes):\n            super().__init__()\n            self.cnn = nn.Sequential(\n                nn.Conv2d(1, 16, 3, padding=1), nn.ReLU(), nn.MaxPool2d(2),\n                nn.Conv2d(16, 32, 3, padding=1), nn.ReLU(), nn.MaxPool2d(2),\n                nn.Conv2d(32, 64, 3, padding=1), nn.ReLU(), nn.AdaptiveAvgPool2d((1, 1)),\n            )\n            self.fc = nn.Linear(64, num_classes)\n\n        def forward(self, x):\n            x = self.cnn(x)\n            x = x.view(x.size(0), -1)\n            return self.fc(x)\n\n#Define paths\nSAVE_PATH = '/kaggle/working/birdclef_cnn.pth' #Newly trained model\nMODEL_PATH = '/kaggle/input/birdclef_stratifiedcnn/pytorch/stratified/2/birdclef_cnnstratified.pth' #Pretrained option\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n#Initialize and load model, this code will favor using the newly trained model but if not detected it will use the pretrained option\nmodel = CNNModel(num_classes=len(unique_labels)).to(device)\n\nif os.path.exists(SAVE_PATH):\n    print(f\"✅ Using newly trained model at {SAVE_PATH}\")\n    model.load_state_dict(torch.load(SAVE_PATH, map_location=device))\nelif os.path.exists(MODEL_PATH):\n    print(f\"⚠️ Newly trained model not found. Using pretrained model at {MODEL_PATH}\")\n    model.load_state_dict(torch.load(MODEL_PATH, map_location=device))\nelse:\n    raise FileNotFoundError(\"❌ No model file found at either SAVE_PATH or MODEL_PATH.\")\n\nfrom sklearn.metrics import accuracy_score, classification_report\n\nmodel.eval()\nval_preds = []\nval_true = []\n\nfor path, label_idx in tqdm(zip(val_paths, val_labels), desc=\"Predicting 20% validation set\", total=len(val_paths)):\n    #Load audio\n    audio, _ = librosa.load(path, sr=SR)\n    chunks = split_audio(audio)\n\n    chunk_probs = []\n\n    for i, chunk in enumerate(chunks):\n        if len(chunk) < CHUNK_LEN * SR:\n            chunk = np.pad(chunk, (0, CHUNK_LEN * SR - len(chunk)))\n\n        mel = to_mel_spectrogram(chunk)\n        mel = normalize_mel(mel)  #Normalization\n        mel_tensor = torch.tensor(mel).unsqueeze(0).unsqueeze(0).float().to(device)\n\n        with torch.no_grad():\n            logits = model(mel_tensor)\n            probs = torch.softmax(logits, dim=1).cpu().numpy().flatten()\n            chunk_probs.append(probs)\n\n    #Aggregate across chunks (e.g. mean of softmax probabilities)\n    avg_probs = np.mean(chunk_probs, axis=0)\n    pred_idx = int(np.argmax(avg_probs))\n\n    val_preds.append(pred_idx)\n    val_true.append(label_idx)\n\n#Evaluation metrics\nacc = accuracy_score(val_true, val_preds)\nprint(f\"\\n✅ Validation accuracy: {acc:.4f}\\n\")\n\nlabels_present = sorted(set(val_true) | set(val_preds))\nprint(\"Classification Report:\")\nprint(classification_report(\n    val_true,\n    val_preds,\n    labels=labels_present,\n    target_names=[unique_labels[i] for i in labels_present],\n    zero_division=0\n))\n\n##Additional output\nfrom collections import Counter\n\ntrain_dist = Counter(train_labels)\nval_dist = Counter(val_labels)\n\nprint(f\"Train classes: {len(train_dist)}\")\nprint(f\"Val classes:   {len(val_dist)}\")\n\n#Visualization\nplt.figure(figsize=(10, 4))\nplt.hist(train_dist.values(), bins=50, alpha=0.6, label='Train')\nplt.hist(val_dist.values(), bins=50, alpha=0.6, label='Validation')\nplt.legend()\nplt.title(\"\")\nplt.xlabel(\"Number of samples per class\")\nplt.ylabel(\"Number of classes\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T14:45:33.676504Z","iopub.execute_input":"2025-05-07T14:45:33.676996Z","iopub.status.idle":"2025-05-07T15:02:42.601094Z","shell.execute_reply.started":"2025-05-07T14:45:33.676951Z","shell.execute_reply":"2025-05-07T15:02:42.600244Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Additional plotting for poster","metadata":{}},{"cell_type":"code","source":"#Visualization\nplt.figure(figsize=(10, 4))\nplt.hist(train_dist.values(), bins=50, alpha=0.6, label='Train')\nplt.hist(val_dist.values(), bins=50, alpha=0.6, label='Validation')\nplt.legend()\nplt.title(\"\")\nplt.xlabel(\"Number of samples per class\")\nplt.ylabel(\"Number of classes\")\nplt.show()\n\nimport os\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\n\n##Visualizing one spectrogram for poster\n#Construct the path to a known file\nlabel_id = \"1194042\"\nfile_name = \"CSA18783.ogg\"\nsample_path = os.path.join(AUDIO_DIR, label_id, file_name)\n\n#Load and process\ny, sr = librosa.load(sample_path, sr=SR)\nchunks = split_audio(y)\n\n#Use the first chunk for visualization\nchunk = chunks[0]\nmel = librosa.feature.melspectrogram(y=chunk, sr=SR, n_mels=N_MELS)\nmel_db = librosa.power_to_db(mel, ref=np.max)\n\n#Plot without normalization (to preserve dB scale)\nplt.figure(figsize=(10, 4))\nlibrosa.display.specshow(mel_db, sr=SR, x_axis='time', y_axis='mel', cmap='magma')\nplt.colorbar(format=\"%+2.0f dB\")\n#plt.title(f\"Mel spectrogram - label {label_id} ({file_name})\")\nplt.title(\"\")\nplt.tight_layout()\nplt.show()\n\n######\nimport pandas as pd\n\n# You already calculated this earlier\nval_acc = acc * 100  # Convert to %\nleaderboard_acc = 78.5  # Example value – replace with your actual leaderboard result\n\n# Create a simple accuracy comparison table\naccuracy_df = pd.DataFrame({\n    'Dataset': ['Validation (20%)', 'Leaderboard'],\n    'Accuracy': [acc, 0.597]  # Replace acc with your actual validation score if needed\n})\n\n# Display nicely\ndisplay(accuracy_df.style.set_properties(**{\n    'text-align': 'center',\n    'font-size': '14px'\n}))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-07T15:02:57.486571Z","iopub.execute_input":"2025-05-07T15:02:57.486888Z","iopub.status.idle":"2025-05-07T15:02:58.891870Z","shell.execute_reply.started":"2025-05-07T15:02:57.486865Z","shell.execute_reply":"2025-05-07T15:02:58.890961Z"}},"outputs":[],"execution_count":null}]}