{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport librosa \nimport matplotlib.pyplot as plt\nimport cv2 as cv\nfrom tqdm import tqdm\nimport math\nimport os\nimport torch \nfrom torch import nn, optim\nfrom torch.utils.data import Dataset, DataLoader, random_split\nimport torchvision\nfrom torchvision.models import efficientnet_v2_m, EfficientNet_V2_M_Weights\nfrom torch.amp import autocast, GradScaler\nfrom torchmetrics.classification import BinaryAUROC\nfrom torchsummary import summary","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:45.048770Z","iopub.execute_input":"2025-05-14T18:53:45.049476Z","iopub.status.idle":"2025-05-14T18:53:49.932494Z","shell.execute_reply.started":"2025-05-14T18:53:45.049443Z","shell.execute_reply":"2025-05-14T18:53:49.931857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\ndevice","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:49.933632Z","iopub.execute_input":"2025-05-14T18:53:49.934061Z","iopub.status.idle":"2025-05-14T18:53:49.994657Z","shell.execute_reply.started":"2025-05-14T18:53:49.934039Z","shell.execute_reply":"2025-05-14T18:53:49.993994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\ntaxo_df = pd.read_csv(\"/kaggle/input/birdclef-2025/taxonomy.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:49.995618Z","iopub.execute_input":"2025-05-14T18:53:49.995871Z","iopub.status.idle":"2025-05-14T18:53:50.120072Z","shell.execute_reply.started":"2025-05-14T18:53:49.995826Z","shell.execute_reply":"2025-05-14T18:53:50.119197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train_df[[\"primary_label\", \"filename\", \"scientific_name\"]].copy()\ndf = df.sort_values(by=\"primary_label\")\nprim_class_ = taxo_df.set_index('primary_label')['class_name'].to_dict()\ndf['label'] = df['primary_label'].astype(str).map(prim_class_) + '_' + df['primary_label'].astype(str)  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:50.121721Z","iopub.execute_input":"2025-05-14T18:53:50.122014Z","iopub.status.idle":"2025-05-14T18:53:50.154975Z","shell.execute_reply.started":"2025-05-14T18:53:50.121990Z","shell.execute_reply":"2025-05-14T18:53:50.154373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:50.155705Z","iopub.execute_input":"2025-05-14T18:53:50.155995Z","iopub.status.idle":"2025-05-14T18:53:50.168512Z","shell.execute_reply.started":"2025-05-14T18:53:50.155965Z","shell.execute_reply":"2025-05-14T18:53:50.167547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Important\nalllabels = set()\nfor sample in df.primary_label:\n    alllabels.add(sample)\n\nalllabels = sorted(list(alllabels))\nn_labels = len(alllabels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:55.326563Z","iopub.execute_input":"2025-05-14T18:53:55.327224Z","iopub.status.idle":"2025-05-14T18:53:55.334967Z","shell.execute_reply.started":"2025-05-14T18:53:55.327196Z","shell.execute_reply":"2025-05-14T18:53:55.334129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    out_dir = \"/kaggle/working/mel_spectrograms/\"\n    in_dir = \"/kaggle/input/birdclef-2025/train_audio/\"\n    fs = 32000\n\n    n_fft = 1024\n    hop_length = 256\n    n_mels = 64\n    fmin = 50\n    fmax = 12000\n\n    target_dur = 5\n    target_shape = (256, 256)\n\nconfig = Config()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:53:57.500438Z","iopub.execute_input":"2025-05-14T18:53:57.501267Z","iopub.status.idle":"2025-05-14T18:53:57.506193Z","shell.execute_reply.started":"2025-05-14T18:53:57.501231Z","shell.execute_reply":"2025-05-14T18:53:57.505319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import IPython.display as ip\n# audio_ = config.in_dir+df.filename[10]\n# label = df.label[10]\n# print(label)\n# ip.Audio(audio_)\n# y, sr = librosa.load(config.in_dir + df.filename[10])\n# mel_spec_test = librosa.feature.melspectrogram(\n#     y=y, \n#     sr=sr,\n#     n_fft=config.n_fft,\n#     hop_length=config.hop_length,\n#     n_mels=config.n_mels,\n#     fmin=config.fmin,\n#     fmax=config.fmax,\n# )\n# mel_db = librosa.power_to_db(mel_spec_test, ref=np.max)\n# librosa.display.specshow(mel_db)\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:54:03.068320Z","iopub.execute_input":"2025-05-14T18:54:03.068833Z","iopub.status.idle":"2025-05-14T18:54:03.072433Z","shell.execute_reply.started":"2025-05-14T18:54:03.068808Z","shell.execute_reply":"2025-05-14T18:54:03.071608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio2mel(audio_):\n    if np.isnan(audio_).any():\n        mean_ = np.nanmean(audio_)\n        audio_ = np.nan_to_num(audio_, nan = mean_)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_,\n        sr=config.fs,\n        n_fft=config.n_fft,\n        hop_length=config.hop_length,\n        n_mels=config.n_mels,\n        fmin=config.fmin,\n        fmax=config.fmax,\n        power=2.0\n    )\n    mel_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_db = (mel_db - mel_db.min()) / (mel_db.max() - mel_db.min() + 1e-8)\n    return mel_db","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.makedirs(config.out_dir, exist_ok=True)\nfor i, row in tqdm(df.iterrows(), total=len(df)):\n    audio_, _  = librosa.load(config.in_dir + row.filename, sr=config.fs)\n    target_samp = int(config.target_dur * config.fs)\n\n    if len(audio_) < target_samp:\n        n_copy = math.ceil(target_samp/len(audio_))\n        if n_copy > 1:\n            audio_ = np.concatenate([audio_] * n_copy)\n    start = max(0, int(len(audio_)/2 - target_samp/2))\n    end = min(len(audio_), start+target_samp)\n\n    audio_ = audio_[start:end]\n    if len(audio_) < target_samp:\n        audio_ = np.pad(audio_, (0, target_samp - len(audio_)), mode='constant')\n    mel_spec = audio2mel(audio_)\n    if mel_spec.shape != config.target_shape:\n        mel_spec = cv.resize(mel_spec, config.target_shape, interpolation=cv.INTER_LINEAR)\n\n    out_path = f\"{config.out_dir}{row.label}_{i}.npy\"\n    np.save(out_path, mel_spec.astype(np.float32))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # !ls ./mel_spectrograms | head -2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdClefDs(Dataset):\n    def __init__(self, mel_dir, alllabels, n_labels):\n        self.mel_dir = mel_dir\n        self.files = [f for f in os.listdir(mel_dir) if f.endswith(\".npy\")]\n        self.alllabels = alllabels # sorted labels list, keep it consistent\n        self.n_labels = n_labels\n    def __len__(self):\n        return len(self.files)\n\n    def __getitem__(self, id_):\n        filename = self.files[id_]\n        filepath = os.path.join(self.mel_dir, filename)\n\n        img = np.load(filepath).astype(np.float32)\n        \n        if img.ndim == 2:\n            img = np.expand_dims(img, axis = 0)\n            \n        img = torch.from_numpy(img)\n        img = img.repeat(3, 1, 1) \n        label_id = self.alllabels.index(filename.split('_')[1])\n        label = torch.zeros(self.n_labels, dtype=torch.float32) # one hot encoding\n        label[label_id] = 1.0\n        return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:54:50.565728Z","iopub.execute_input":"2025-05-14T18:54:50.566359Z","iopub.status.idle":"2025-05-14T18:54:50.572776Z","shell.execute_reply.started":"2025-05-14T18:54:50.566329Z","shell.execute_reply":"2025-05-14T18:54:50.571972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BirdClef = BirdClefDs(config.out_dir, alllabels, n_labels)\n\nfor i, (sample, label) in enumerate(BirdClef):\n    print(f\"{i}: {sample.size()}, {label}\")\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:54:53.779571Z","iopub.execute_input":"2025-05-14T18:54:53.780389Z","iopub.status.idle":"2025-05-14T18:54:53.807176Z","shell.execute_reply.started":"2025-05-14T18:54:53.780363Z","shell.execute_reply":"2025-05-14T18:54:53.806474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_size = int(0.8 * len(BirdClef))\nval_size = len(BirdClef) - train_size\nBirdClef_train, BirdClef_val = random_split(BirdClef, [train_size, val_size])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:54:56.404703Z","iopub.execute_input":"2025-05-14T18:54:56.405009Z","iopub.status.idle":"2025-05-14T18:54:56.411028Z","shell.execute_reply.started":"2025-05-14T18:54:56.404987Z","shell.execute_reply":"2025-05-14T18:54:56.410369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size=8\ntrain_ = DataLoader(BirdClef_train, batch_size=batch_size, shuffle=True, num_workers=4)\nval_ = DataLoader(BirdClef_val, batch_size=batch_size, shuffle=False, num_workers=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:54:57.311676Z","iopub.execute_input":"2025-05-14T18:54:57.312237Z","iopub.status.idle":"2025-05-14T18:54:57.316671Z","shell.execute_reply.started":"2025-05-14T18:54:57.312213Z","shell.execute_reply":"2025-05-14T18:54:57.315731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for x, y in train_:\n#     print(f\"{x.shape} : {y.shape}\")\n#     print(f\"{x} : {y}\")\n#     break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Model_01(nn.Module):\n    def __init__(self, n_labels):\n        super().__init__()\n        base = efficientnet_v2_m(weights=EfficientNet_V2_M_Weights.DEFAULT)\n        \n        self.features = base.features\n        self.avg_pool = base.avgpool\n        self.classifier = nn.Sequential(\n            nn.Dropout(0.4),\n            nn.Linear(base.classifier[1].in_features, n_labels)\n        )\n    def forward(self, x):\n        x = self.features(x)\n        x = self.avg_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.classifier(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:55:00.086188Z","iopub.execute_input":"2025-05-14T18:55:00.086969Z","iopub.status.idle":"2025-05-14T18:55:00.093432Z","shell.execute_reply.started":"2025-05-14T18:55:00.086933Z","shell.execute_reply":"2025-05-14T18:55:00.092552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Model_01(n_labels).to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:55:00.660167Z","iopub.execute_input":"2025-05-14T18:55:00.660459Z","iopub.status.idle":"2025-05-14T18:55:02.110569Z","shell.execute_reply.started":"2025-05-14T18:55:00.660438Z","shell.execute_reply":"2025-05-14T18:55:02.109987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summary(model, input_size=(3, 256, 256))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lr = 1e-4\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=lr, weight_decay=1e-4)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:55:08.827211Z","iopub.execute_input":"2025-05-14T18:55:08.827535Z","iopub.status.idle":"2025-05-14T18:55:08.835302Z","shell.execute_reply.started":"2025-05-14T18:55:08.827513Z","shell.execute_reply":"2025-05-14T18:55:08.834189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# output = model(x.to(device))\n# output.shape, y.shape\n# auroc = BinaryAUROC()\n# auc_score = auroc(output.cpu(), y)\n\n# print(f\"AUROC score: {auc_score}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:56:21.621376Z","iopub.execute_input":"2025-05-14T18:56:21.621675Z","iopub.status.idle":"2025-05-14T18:56:21.625274Z","shell.execute_reply.started":"2025-05-14T18:56:21.621654Z","shell.execute_reply":"2025-05-14T18:56:21.624363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(output[0].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:56:23.637242Z","iopub.execute_input":"2025-05-14T18:56:23.637551Z","iopub.status.idle":"2025-05-14T18:56:23.641193Z","shell.execute_reply.started":"2025-05-14T18:56:23.637528Z","shell.execute_reply":"2025-05-14T18:56:23.640321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train(model, train_, criterion, optimizer, scheduler, n_epochs=10):\n    losses = []\n    auroces = []\n    scaler = GradScaler(device=device)\n    metric = BinaryAUROC().to(device)\n    \n    for epoch in range(n_epochs):\n        model.train()\n        running_loss = 0.0\n        \n        for sample, label in tqdm(train_, total=len(train_)):\n            sample = sample.to(device)\n            label = label.to(device)\n    \n            optimizer.zero_grad()\n            \n            with autocast(device_type=device):\n                yhat = model(sample)\n                loss = criterion(yhat, label)\n                \n            scaler.scale(loss).backward() # loss.backward()\n            scaler.step(optimizer) # optimizer.step()\n            scaler.update()\n                        \n            running_loss += loss.item() * sample.size(0)\n            metric.update(yhat.sigmoid(), label)\n        \n        auroc = metric.compute().item()\n        epoch_loss = running_loss / len(train_.dataset)\n\n        losses.append(epoch_loss)\n        auroces.append(auroc)\n        metric.reset()\n        \n        print(f'Epoch {epoch+1}/{n_epochs} - Loss: {epoch_loss:.4f}, ROC: {auroc:.4f}')\n        scheduler.step(epoch_loss)\n        \n    return losses, auroces\n\nlosses, auroces = train(model, train_, criterion, optimizer, scheduler)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T18:56:34.680883Z","iopub.execute_input":"2025-05-14T18:56:34.681561Z","iopub.status.idle":"2025-05-14T20:12:42.741648Z","shell.execute_reply.started":"2025-05-14T18:56:34.681536Z","shell.execute_reply":"2025-05-14T20:12:42.740624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(losses, auroces)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T20:12:42.744597Z","iopub.execute_input":"2025-05-14T20:12:42.744826Z","iopub.status.idle":"2025-05-14T20:12:42.749254Z","shell.execute_reply.started":"2025-05-14T20:12:42.744807Z","shell.execute_reply":"2025-05-14T20:12:42.748473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(losses)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T20:12:42.750201Z","iopub.execute_input":"2025-05-14T20:12:42.750460Z","iopub.status.idle":"2025-05-14T20:12:42.905723Z","shell.execute_reply.started":"2025-05-14T20:12:42.750438Z","shell.execute_reply":"2025-05-14T20:12:42.905062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(auroces)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T20:12:42.907491Z","iopub.execute_input":"2025-05-14T20:12:42.907710Z","iopub.status.idle":"2025-05-14T20:12:43.026008Z","shell.execute_reply.started":"2025-05-14T20:12:42.907695Z","shell.execute_reply":"2025-05-14T20:12:43.025375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate(model, val_, criterion):\n    model.eval()\n    val_loss = 0.0\n    metric = BinaryAUROC().to(device)\n    with torch.no_grad():\n        for sample, label in tqdm(val_):\n            sample = sample.to(device)\n            label = label.to(device) \n            \n            with autocast(device_type=device):\n                y_pred = model(sample)\n                loss = criterion(y_pred.sigmoid(), label)\n                \n            metric.update(y_pred.sigmoid(), label)    \n            val_loss += loss.item() *  sample.size(0)\n            \n    auroc = metric.compute().item()     \n    val_loss /= len(val_.dataset)\n    return val_loss, auroc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T20:12:43.026642Z","iopub.execute_input":"2025-05-14T20:12:43.026827Z","iopub.status.idle":"2025-05-14T20:12:43.032448Z","shell.execute_reply.started":"2025-05-14T20:12:43.026812Z","shell.execute_reply":"2025-05-14T20:12:43.031560Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_loss, auroc = evaluate(model, val_, criterion)\nprint(f\"Validation Loss: {val_loss:.4f} | ROC: {auroc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T20:12:43.033201Z","iopub.execute_input":"2025-05-14T20:12:43.033452Z","iopub.status.idle":"2025-05-14T20:13:16.426797Z","shell.execute_reply.started":"2025-05-14T20:12:43.033430Z","shell.execute_reply":"2025-05-14T20:13:16.425756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(model.state_dict(), \"/kaggle/working/BirdClefmodel01.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T20:13:16.427752Z","iopub.execute_input":"2025-05-14T20:13:16.428004Z","iopub.status.idle":"2025-05-14T20:13:16.787329Z","shell.execute_reply.started":"2025-05-14T20:13:16.427980Z","shell.execute_reply":"2025-05-14T20:13:16.786733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}