{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11860649,"sourceType":"datasetVersion","datasetId":7452919}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n!pip install timm\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport timm\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:47:17.338835Z","iopub.execute_input":"2025-05-28T07:47:17.339060Z","iopub.status.idle":"2025-05-28T07:50:20.099386Z","shell.execute_reply.started":"2025-05-28T07:47:17.339035Z","shell.execute_reply":"2025-05-28T07:50:20.098538Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import librosa\nimport torch\nimport torchaudio\nimport matplotlib.pyplot as plt\nimport ast\nimport numpy as np\n\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.metrics import roc_auc_score\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:20.100923Z","iopub.execute_input":"2025-05-28T07:50:20.101359Z","iopub.status.idle":"2025-05-28T07:50:21.961459Z","shell.execute_reply.started":"2025-05-28T07:50:20.101334Z","shell.execute_reply":"2025-05-28T07:50:21.960723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Constants\nSAMPLE_RATE = 32000\nDURATION = 5  # seconds\nN_MELS = 128\nHOP_LENGTH = 512\nN_FFT = 2048\nNUM_CLASSES = 206\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:21.962163Z","iopub.execute_input":"2025-05-28T07:50:21.962517Z","iopub.status.idle":"2025-05-28T07:50:22.047834Z","shell.execute_reply.started":"2025-05-28T07:50:21.962500Z","shell.execute_reply":"2025-05-28T07:50:22.047039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_audio(file_path, duration=DURATION, sr=SAMPLE_RATE):\n    y, _ = librosa.load(file_path, sr=sr, mono=True)\n    if len(y) < sr * duration:\n        y = np.pad(y, (0, sr * duration - len(y)))\n    else:\n        y = y[:sr * duration]\n    return y\n\ndef audio_to_logmel(y, sr=SAMPLE_RATE):\n    mel = librosa.feature.melspectrogram(y=y, sr=sr, n_fft=N_FFT, hop_length=HOP_LENGTH, n_mels=N_MELS)\n    logmel = librosa.power_to_db(mel)\n   \n    logmel = (logmel - logmel.mean(axis=1, keepdims=True)) / (logmel.std(axis=1, keepdims=True) + 1e-6)\n    return logmel\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:22.048802Z","iopub.execute_input":"2025-05-28T07:50:22.049576Z","iopub.status.idle":"2025-05-28T07:50:22.066985Z","shell.execute_reply.started":"2025-05-28T07:50:22.049555Z","shell.execute_reply":"2025-05-28T07:50:22.066252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EfficientNetFrozen(nn.Module):\n    def __init__(self, model_name='efficientnet_b0', n_classes=206):\n        super().__init__()\n        self.backbone = timm.create_model(model_name, pretrained=True, in_chans=1, num_classes=0)\n        \n        # Freeze backbone weights\n        for param in self.backbone.parameters():\n            param.requires_grad = False\n\n        self.classifier = nn.Linear(self.backbone.num_features, n_classes)\n\n    def forward(self, x):\n        with torch.no_grad():  # Optional: safer if frozen\n            features = self.backbone(x)\n        logits = self.classifier(features)\n        return logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:22.067755Z","iopub.execute_input":"2025-05-28T07:50:22.067977Z","iopub.status.idle":"2025-05-28T07:50:22.089665Z","shell.execute_reply.started":"2025-05-28T07:50:22.067962Z","shell.execute_reply":"2025-05-28T07:50:22.089198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model(model, train_dl, val_dl, optimizer, criterion, num_epochs=10, patience=3, fold=0):\n    model.to(DEVICE)\n\n    best_auc = 0\n    epochs_since_improvement = 0\n    best_model_state = None\n\n    for epoch in range(num_epochs):\n        model.train()\n        train_loss = 0\n        for xb, yb in tqdm(train_dl, desc=f\"Epoch {epoch+1} - Training\"):\n            xb, yb = xb.to(DEVICE), yb.to(DEVICE)\n            optimizer.zero_grad()\n            outputs = model(xb)\n            loss = criterion(outputs, yb)\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item()\n\n        model.eval()\n        val_loss = 0\n        all_preds = []\n        all_targets = []\n        with torch.no_grad():\n            for xb, yb in tqdm(val_dl, desc=f\"Epoch {epoch+1} - Validation\"):\n                xb, yb = xb.to(DEVICE), yb.to(DEVICE)\n                outputs = model(xb)\n                loss = criterion(outputs, yb)\n                val_loss += loss.item()\n                all_preds.append(torch.sigmoid(outputs).cpu().numpy())\n                all_targets.append(yb.cpu().numpy())\n\n        all_preds = np.vstack(all_preds)\n        all_targets = np.vstack(all_targets)\n\n        aucs = []\n        for i in range(all_targets.shape[1]):\n            if np.sum(all_targets[:, i]) > 0:\n                auc = roc_auc_score(all_targets[:, i], all_preds[:, i])\n                aucs.append(auc)\n        macro_auc = np.mean(aucs)\n\n        print(f\"Epoch {epoch+1}: Train Loss = {train_loss/len(train_dl):.4f} | Val Loss = {val_loss/len(val_dl):.4f} | Val Macro ROC-AUC = {macro_auc:.4f}\")\n\n        # Early stopping logic\n        if macro_auc > best_auc:\n            best_auc = macro_auc\n            epochs_since_improvement = 0\n            best_model_path = f\"efficientnet_b0_frozen_fold{fold}_best.pth\"\n            torch.save(model.state_dict(), best_model_path)\n        else:\n            epochs_since_improvement += 1\n            if epochs_since_improvement >= patience:\n                print(f\"⏹️ Early stopping triggered after {epoch+1} epochs.\")\n                break\n\n    # Load best model state before returning\n    if best_model_state:\n        model.load_state_dict(best_model_state)\n\n    return best_auc\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:22.090275Z","iopub.execute_input":"2025-05-28T07:50:22.090434Z","iopub.status.idle":"2025-05-28T07:50:22.110649Z","shell.execute_reply.started":"2025-05-28T07:50:22.090421Z","shell.execute_reply":"2025-05-28T07:50:22.109976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === Setup Paths ===\nDATA_PATH = \"/kaggle/input/birdclef-2025\"\nAUDIO_PATH = os.path.join(DATA_PATH, \"train_audio\")\n\n# === Load Data ===\ndf = pd.read_csv(os.path.join(DATA_PATH, \"train.csv\"))\n\n# === Combine Primary + Secondary Labels ===\ndef combine_labels(row):\n    if pd.isna(row['secondary_labels']):\n        secondary = []\n    else:\n        secondary = ast.literal_eval(row['secondary_labels'])\n        # remove empty strings from parsed list\n        secondary = [s for s in secondary if s.strip() != '']\n    return [row['primary_label']] + secondary\n\ndf['labels'] = df.apply(combine_labels, axis=1)\n\n# === Fit MultiLabelBinarizer ===\nmlb = MultiLabelBinarizer()\nmlb.fit(df['labels'])  # 🔥 Only once, on full label set\n\nNUM_CLASSES = len(mlb.classes_)\n\n# === Define Dataset ===\nclass BirdDataset(Dataset):\n    def __init__(self, df, audio_dir, mlb):\n        self.df = df.reset_index(drop=True)\n        self.audio_dir = audio_dir\n        self.mlb = mlb\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        y = load_audio(os.path.join(self.audio_dir, row['filename']))\n        melspec = audio_to_logmel(y)\n        melspec = torch.tensor(melspec).unsqueeze(0).float()\n\n        label_list = row['labels']  # Already combined\n        label_vec = self.mlb.transform([label_list])[0]\n\n        # ✅ Debug: Confirm shape\n        if idx == 0:\n            print(\"Label list:\", label_list)\n            print(\"Transformed label shape:\", label_vec.shape)\n\n        return melspec, torch.tensor(label_vec).float()\nNUM_FOLDS = 3\nskf = StratifiedKFold(n_splits=NUM_FOLDS, shuffle=True, random_state=42)\n\nfold_aucs = []\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:22.112407Z","iopub.execute_input":"2025-05-28T07:50:22.112608Z","iopub.status.idle":"2025-05-28T07:50:22.917278Z","shell.execute_reply.started":"2025-05-28T07:50:22.112593Z","shell.execute_reply":"2025-05-28T07:50:22.916627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for fold, (train_idx, val_idx) in enumerate(skf.split(df, df['primary_label'])):\n    print(f\"\\n===== Fold {fold + 1} / {NUM_FOLDS} =====\")\n\n    train_df = df.iloc[train_idx].reset_index(drop=True)\n    val_df = df.iloc[val_idx].reset_index(drop=True)\n\n    train_ds = BirdDataset(train_df, AUDIO_PATH, mlb)\n    val_ds = BirdDataset(val_df, AUDIO_PATH, mlb)\n\n    train_dl = DataLoader(train_ds, batch_size=16, shuffle=True, num_workers=2)\n    val_dl = DataLoader(val_ds, batch_size=16, shuffle=False, num_workers=2)\n\n    model = EfficientNetFrozen(model_name='efficientnet_b0', n_classes=NUM_CLASSES).to(DEVICE)\n    optimizer = torch.optim.Adam(model.classifier.parameters(), lr=1e-3)\n    criterion = nn.BCEWithLogitsLoss()\n \n    fold_auc = train_model(model, train_dl, val_dl, optimizer, criterion, num_epochs=10, patience=3, fold=fold)\n    fold_aucs.append(fold_auc)\n\n    torch.save(model.state_dict(), f\"efficientnet_b0_frozen_fold{fold}.pth\")\n\nprint(f\"\\n✅ Average ROC-AUC across {NUM_FOLDS} folds: {np.mean(fold_aucs):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T07:50:22.917917Z","iopub.execute_input":"2025-05-28T07:50:22.918134Z"}},"outputs":[],"execution_count":null}]}