{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":364095,"sourceType":"modelInstanceVersion","modelInstanceId":302126,"modelId":322625},{"sourceId":373585,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":308961,"modelId":329363}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import the required libraries\n**Justification**\n* Os is used to work with the file and directory paths\n* Torch is used for the CNN operations\n* Librosa is used to load the .ogg files, compute the mel spectrograms and convert the scale into a decibel scale.\n* Numpy is used for the numerical operations and processing\n* Pandas is used for creating the final submission csv file","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport librosa\nimport numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T01:18:15.253783Z","iopub.execute_input":"2025-05-05T01:18:15.254047Z","iopub.status.idle":"2025-05-05T01:18:17.705076Z","shell.execute_reply.started":"2025-05-05T01:18:15.254026Z","shell.execute_reply":"2025-05-05T01:18:17.703835Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Parameter settings\n* This part of the code sets directories and defines some parameters which are used later in the script","metadata":{}},{"cell_type":"code","source":"#Parameters\nTEST_DIR = '/kaggle/input/birdclef-2025/test_soundscapes/'\nMODEL_PATH = '/kaggle/input/birdclef_stratifiedcnn/pytorch/stratified/2/birdclef_cnnstratified.pth'#The model with stratification will automatically be loaded\nSUBMIT_PATH = '/kaggle/working/submission.csv'\nSR = 32000 #Sampling rate 32K hz\nCHUNK_LEN = 5 #5 seconds\nN_MELS = 128 #the amount of mel frequency bands used to convert audios to spectrogram\nBATCH_SIZE = 32 #Choose batch size\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu') #Define when to use gpu or cpu","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-05T01:18:24.077230Z","iopub.execute_input":"2025-05-05T01:18:24.077594Z","iopub.status.idle":"2025-05-05T01:18:24.084208Z","shell.execute_reply.started":"2025-05-05T01:18:24.077569Z","shell.execute_reply":"2025-05-05T01:18:24.083134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Label mapping\n* This part loads the labels in the specific order that is used in the training script to accurately map the predictions later in the script","metadata":{}},{"cell_type":"code","source":"#Load label mapping from training\nunique_labels = [\n    \"1139490\", \"1192948\", \"1194042\", \"126247\", \"1346504\", \"134933\", \"135045\", \"1462711\", \"1462737\", \"1564122\",\n    \"21038\", \"21116\", \"21211\", \"22333\", \"22973\", \"22976\", \"24272\", \"24292\", \"24322\", \"41663\",\n    \"41778\", \"41970\", \"42007\", \"42087\", \"42113\", \"46010\", \"47067\", \"476537\", \"476538\", \"48124\",\n    \"50186\", \"517119\", \"523060\", \"528041\", \"52884\", \"548639\", \"555086\", \"555142\", \"566513\", \"64862\",\n    \"65336\", \"65344\", \"65349\", \"65373\", \"65419\", \"65448\", \"65547\", \"65962\", \"66016\", \"66531\",\n    \"66578\", \"66893\", \"67082\", \"67252\", \"714022\", \"715170\", \"787625\", \"81930\", \"868458\", \"963335\",\n    \"amakin1\", \"amekes\", \"ampkin1\", \"anhing\", \"babwar\", \"bafibi1\", \"banana\", \"baymac\", \"bbwduc\", \"bicwre1\",\n    \"bkcdon\", \"bkmtou1\", \"blbgra1\", \"blbwre1\", \"blcant4\", \"blchaw1\", \"blcjay1\", \"blctit1\", \"blhpar1\", \"blkvul\",\n    \"bobfly1\", \"bobher1\", \"brtpar1\", \"bubcur1\", \"bubwre1\", \"bucmot3\", \"bugtan\", \"butsal1\", \"cargra1\", \"cattyr\",\n    \"chbant1\", \"chfmac1\", \"cinbec1\", \"cocher1\", \"cocwoo1\", \"colara1\", \"colcha1\", \"compau\", \"compot1\", \"cotfly1\",\n    \"crbtan1\", \"crcwoo1\", \"crebob1\", \"cregua1\", \"creoro1\", \"eardov1\", \"fotfly\", \"gohman1\", \"grasal4\", \"grbhaw1\",\n    \"greani1\", \"greegr\", \"greibi1\", \"grekis\", \"grepot1\", \"gretin1\", \"grnkin\", \"grysee1\", \"gybmar\", \"gycwor1\",\n    \"labter1\", \"laufal1\", \"leagre\", \"linwoo1\", \"littin1\", \"mastit1\", \"neocor\", \"norscr1\", \"olipic1\", \"orcpar\",\n    \"palhor2\", \"paltan1\", \"pavpig2\", \"piepuf1\", \"pirfly1\", \"piwtyr1\", \"plbwoo1\", \"plctan1\", \"plukit1\", \"purgal2\",\n    \"ragmac1\", \"rebbla1\", \"recwoo1\", \"rinkin1\", \"roahaw\", \"rosspo1\", \"royfly1\", \"rtlhum\", \"rubsee1\", \"rufmot1\",\n    \"rugdov\", \"rumfly1\", \"ruther1\", \"rutjac1\", \"rutpuf1\", \"saffin\", \"sahpar1\", \"savhaw1\", \"secfly1\", \"shghum1\",\n    \"shtfly1\", \"smbani\", \"snoegr\", \"sobtyr1\", \"socfly1\", \"solsan\", \"soulap1\", \"spbwoo1\", \"speowl1\", \"spepar1\",\n    \"srwswa1\", \"stbwoo2\", \"strcuc1\", \"strfly1\", \"strher\", \"strowl1\", \"tbsfin1\", \"thbeup1\", \"thlsch3\", \"trokin\",\n    \"tropar\", \"trsowl\", \"turvul\", \"verfly\", \"watjac1\", \"wbwwre1\", \"whbant1\", \"whbman1\", \"whfant1\", \"whmtyr1\",\n    \"whtdov\", \"whttro1\", \"whwswa1\", \"woosto\", \"y00678\", \"yebela1\", \"yebfly1\", \"yebsee1\", \"yecspi2\", \"yectyr1\",\n    \"yehbla2\", \"yehcar1\", \"yelori1\", \"yeofly1\", \"yercac1\", \"ywcpar\"\n]\n\nidx2label = {i: label for i, label in enumerate(unique_labels)}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Helper functions + loading CNN model structure\n* This part loads two helper functions:\n* The first helper function splits the audios into 5 second chunks\n* The second helper function convert and normalizes the chunks into mel spectrograms\n* The same CNN architecture is defined as in the training script and the model is loaded from the MODEL_PATH","metadata":{}},{"cell_type":"code","source":"#Helper functions\ndef split_audio(audio, sr=SR, chunk_length=CHUNK_LEN):\n    samples_per_chunk = chunk_length * sr\n    return [audio[i:i+samples_per_chunk] for i in range(0, len(audio), samples_per_chunk)]\n\ndef to_mel_spectrogram(chunk):\n    mel = librosa.feature.melspectrogram(y=chunk, sr=SR, n_mels=N_MELS)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n    mel_db -= mel_db.min()\n    mel_db /= mel_db.max()\n    return mel_db\n\n#Defining model architecture\nclass CNNModel(nn.Module):\n    def __init__(self, num_classes):\n        super().__init__()\n        self.cnn = nn.Sequential(\n            nn.Conv2d(1, 16, 3, padding=1), nn.ReLU(), nn.MaxPool2d(2),\n            nn.Conv2d(16, 32, 3, padding=1), nn.ReLU(), nn.MaxPool2d(2),\n            nn.Conv2d(32, 64, 3, padding=1), nn.ReLU(), nn.AdaptiveAvgPool2d((1, 1)),\n        )\n        self.fc = nn.Linear(64, num_classes)\n\n    def forward(self, x):\n        x = self.cnn(x)\n        x = x.view(x.size(0), -1)\n        return self.fc(x)\n\n#Loading model obtained from training script\nmodel = CNNModel(num_classes=len(unique_labels)).to(device)\nmodel.load_state_dict(torch.load(MODEL_PATH, map_location=device))\nmodel.eval()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Processing test data and creating submission dataframe\n* The final part of the code processes the test/validation data\n* First the validation set is located and filtered\n* The validation audios are splitted into chunks and converted to mel spectrograms\n* The model predicts and the output is formatted and saved in the desired competition format","metadata":{}},{"cell_type":"code","source":"#Iterates over test files, filtering for .ogg\nrows = []\nfor fname in sorted(os.listdir(TEST_DIR)):\n    if not fname.endswith('.ogg'):\n        continue\n        \n    #Loads audio files and splits into chunks\n    path = os.path.join(TEST_DIR, fname)\n    audio, _ = librosa.load(path, sr=SR)\n    chunks = split_audio(audio)\n    \n    #Extracting the soundscape ID's\n    soundscape_id = fname.split('_')[1].split('.')[0]\n\n    #Pad audio if audio is too short\n    for i, chunk in enumerate(chunks):\n        if len(chunk) < CHUNK_LEN * SR:\n            chunk = np.pad(chunk, (0, CHUNK_LEN * SR - len(chunk)))\n\n        #Convert to mel_spectrogram\n        mel = to_mel_spectrogram(chunk)\n        mel_tensor = torch.tensor(mel).unsqueeze(0).unsqueeze(0).float().to(device)\n\n        #Run the model prediction\n        with torch.no_grad():\n            logits = model(mel_tensor)\n            probs = torch.softmax(logits, dim=1).cpu().numpy().flatten()\n\n        #Row_id formatting: soundscape_[soundscape_id]_[end_time]\n        end_time = (i + 1) * CHUNK_LEN\n        row_id = f\"soundscape_{soundscape_id}_{end_time}\"\n\n        row = [row_id] + probs.tolist()\n        rows.append(row)\n\n#Create the submission dataframe and save this \ncolumns = ['row_id'] + unique_labels\ndf = pd.DataFrame(rows, columns=columns)\ndf.to_csv(SUBMIT_PATH, index=False)\nprint(f\"✅ Submission saved to {SUBMIT_PATH}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}