{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":11498503,"sourceType":"datasetVersion","datasetId":7201120},{"sourceId":349410,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":291770,"modelId":312431}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:22:45.970898Z","iopub.execute_input":"2025-04-21T10:22:45.97109Z","iopub.status.idle":"2025-04-21T10:22:46.143537Z","shell.execute_reply.started":"2025-04-21T10:22:45.971073Z","shell.execute_reply":"2025-04-21T10:22:46.142909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q efficientnet-pytorch ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:22:46.144414Z","iopub.execute_input":"2025-04-21T10:22:46.144616Z","iopub.status.idle":"2025-04-21T10:24:05.917714Z","shell.execute_reply.started":"2025-04-21T10:22:46.144597Z","shell.execute_reply":"2025-04-21T10:24:05.917041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport torch\nimport numpy as np\nimport pandas as pd \nfrom torch.utils.data import Dataset\nfrom torchvision import transforms\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, random_split\nfrom torch.optim import AdamW\nfrom torch.optim.lr_scheduler import CosineAnnealingLR\nfrom glob import glob\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:24:05.919726Z","iopub.execute_input":"2025-04-21T10:24:05.919949Z","iopub.status.idle":"2025-04-21T10:24:15.437614Z","shell.execute_reply.started":"2025-04-21T10:24:05.919931Z","shell.execute_reply":"2025-04-21T10:24:15.437069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\nseed_everything()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:24:15.438378Z","iopub.execute_input":"2025-04-21T10:24:15.438695Z","iopub.status.idle":"2025-04-21T10:24:15.448282Z","shell.execute_reply.started":"2025-04-21T10:24:15.438676Z","shell.execute_reply":"2025-04-21T10:24:15.447623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2021-mel-spectograms/train_metadata_extended.csv')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T18:14:49.147461Z","iopub.execute_input":"2025-04-20T18:14:49.147677Z","iopub.status.idle":"2025-04-20T18:14:49.7317Z","shell.execute_reply.started":"2025-04-20T18:14:49.14766Z","shell.execute_reply":"2025-04-20T18:14:49.730902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdSpectrogramDataset(Dataset):\n    def __init__(self, root_dir, class_map, transform=None):\n        self.root_dir = root_dir\n        self.class_map = class_map  # dict: {bird_name: class_idx}\n        self.transform = transform\n        \n        self.samples = []\n        for bird_name in os.listdir(root_dir):\n            bird_path = os.path.join(root_dir, bird_name)\n            if os.path.isdir(bird_path):\n                for fname in os.listdir(bird_path):\n                    if fname.endswith('.npy'):\n                        self.samples.append((os.path.join(bird_path, fname), bird_name))\n        \n    def __len__(self):\n        return len(self.samples)\n    \n    def __getitem__(self, idx):\n        path, bird_name = self.samples[idx]\n        spec_stack = np.load(path)  # [12, 128, 321]\n        \n        # Select a random 5s slice\n        slice_idx = random.randint(0, spec_stack.shape[0] - 1)\n        spec = spec_stack[slice_idx]  # shape [128, 321]\n        \n        # Normalize and convert to 3-channel for EfficientNet\n        spec = torch.tensor(spec).float().unsqueeze(0)  # [1, 128, 321]\n        spec = spec.repeat(3, 1, 1)  # [3, 128, 321]\n        \n        if self.transform:\n            spec = self.transform(spec)\n\n        # TODO: CHANGE TO MULTIPLE LABELS FOR SINGLE AND PRIMARY \n        label = torch.zeros(len(self.class_map))\n        label[self.class_map[bird_name]] = 1.0\n        \n        return spec, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T18:14:49.733478Z","iopub.execute_input":"2025-04-20T18:14:49.733706Z","iopub.status.idle":"2025-04-20T18:14:49.740465Z","shell.execute_reply.started":"2025-04-20T18:14:49.73368Z","shell.execute_reply":"2025-04-20T18:14:49.739767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\n\nwith open('/kaggle/input/birdclef-2021-mel-spectograms/LABEL_IDS.json') as f:\n    class_map = json.load(f) \n\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n])\n\n\ndataset = BirdSpectrogramDataset('/kaggle/input/birdclef-2021-mel-spectograms/mel_spectograms/mel_spectograms', class_map, transform)\n\n# Split dataset\ntrain_size = int(0.85 * len(dataset))\nval_size = len(dataset) - train_size\ntrain_ds, val_ds = random_split(dataset, [train_size, val_size])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T18:14:49.741254Z","iopub.execute_input":"2025-04-20T18:14:49.741535Z","iopub.status.idle":"2025-04-20T18:14:52.388178Z","shell.execute_reply.started":"2025-04-20T18:14:49.741509Z","shell.execute_reply":"2025-04-20T18:14:52.387367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 96\n\ntrain_loader = DataLoader(train_ds, batch_size=batch_size, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_ds, batch_size=batch_size, shuffle=False, num_workers=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T18:14:52.389056Z","iopub.execute_input":"2025-04-20T18:14:52.389329Z","iopub.status.idle":"2025-04-20T18:14:52.393565Z","shell.execute_reply.started":"2025-04-20T18:14:52.389296Z","shell.execute_reply":"2025-04-20T18:14:52.392825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\nNUM_CLASSES = 397 \n# Model\nmodel = EfficientNet.from_pretrained('efficientnet-b0', num_classes=NUM_CLASSES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T18:14:52.394333Z","iopub.execute_input":"2025-04-20T18:14:52.394789Z","iopub.status.idle":"2025-04-20T18:14:52.508229Z","shell.execute_reply.started":"2025-04-20T18:14:52.394772Z","shell.execute_reply":"2025-04-20T18:14:52.507459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm.notebook import tqdm  # Import tqdm for notebooks\n\n# Configs\nepochs = 10  # between 7–12\nlr = 2e-3\nlog_every = 2\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(\"DEVICE: \", device)\n\n# Loss, optimizer, scheduler\ncriterion = nn.CrossEntropyLoss()\noptimizer = AdamW(model.parameters(), lr=lr)\nscheduler = CosineAnnealingLR(optimizer, T_max=epochs)\n\nmodel.to(device)\n\nfor epoch in range(epochs):\n    model.train()\n    train_loss = 0.0\n\n    # Wrapping the train_loader with tqdm for progress bar (notebook version)\n    with tqdm(train_loader, unit='batch', desc=f\"Epoch {epoch+1}/{epochs} Train\") as train_bar:\n        for inputs, targets in train_bar:\n            inputs, targets = inputs.to(device), targets.to(device)\n\n            optimizer.zero_grad()\n            outputs = model(inputs)\n            loss = criterion(outputs, targets.argmax(dim=1))  # CrossEntropyLoss expects class index\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item() * inputs.size(0)\n            \n            # Update the tqdm description with current training loss\n            avg_loss = train_loss / ((train_bar.n + 1) * inputs.size(0))  # Correctly calculate average loss\n            train_bar.set_postfix(loss=avg_loss)\n\n    scheduler.step()\n\n    if (epoch + 1) % log_every == 0 or epoch == 0:\n        avg_train_loss = train_loss / len(train_loader.dataset)\n        print(f\"[Epoch {epoch+1}/{epochs}] Train Loss: {avg_train_loss:.4f}\")\n\n    # Validation step with tqdm (notebook version)\n    model.eval()\n    val_loss = 0.0\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        with tqdm(val_loader, unit='batch', desc=f\"Epoch {epoch+1}/{epochs} Val\") as val_bar:\n            for inputs, targets in val_bar:\n                inputs, targets = inputs.to(device), targets.to(device)\n                outputs = model(inputs)\n\n                loss = criterion(outputs, targets.argmax(dim=1))\n                val_loss += loss.item() * inputs.size(0)\n\n                preds = outputs.argmax(dim=1)\n                labels = targets.argmax(dim=1)\n                correct += (preds == labels).sum().item()\n                total += inputs.size(0)\n\n                # Update the tqdm description with current validation loss and accuracy\n                avg_val_loss = val_loss / ((val_bar.n + 1) * inputs.size(0))  # Correctly calculate average val loss\n                val_acc = correct / total\n                val_bar.set_postfix(loss=avg_val_loss, accuracy=val_acc)\n\n    avg_val_loss = val_loss / len(val_loader.dataset)\n    val_acc = correct / total\n    print(f\"→ Validation Loss: {avg_val_loss:.4f} | Accuracy: {val_acc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T18:14:52.509141Z","iopub.execute_input":"2025-04-20T18:14:52.509405Z","iopub.status.idle":"2025-04-20T18:56:52.696355Z","shell.execute_reply.started":"2025-04-20T18:14:52.509387Z","shell.execute_reply":"2025-04-20T18:56:52.695446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(model.state_dict(), 'best_model_weights.pth')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-20T19:00:10.732264Z","iopub.execute_input":"2025-04-20T19:00:10.732999Z","iopub.status.idle":"2025-04-20T19:00:10.820438Z","shell.execute_reply.started":"2025-04-20T19:00:10.732966Z","shell.execute_reply":"2025-04-20T19:00:10.819873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model\n\nWe use the EfficientNetB0 architecture, which should be decently fast for training and, more importantly, re-training with new albels.\n\nTODO: \n- Horizontal cutmix data augmentation\n- CELoss vs BCELoss ? CELoss + Softmax at training, Sigmoid at inference\n- Other things suggested in https://www.kaggle.com/competitions/birdclef-2024/discussion/512197 (min ensembling to use uncertainty)\n- need to use secondary lbales also\n- Don't use softmax at inference, use sigmoid -- and then do some\n- dont stop yet, instead stop with val loss plateau. LR schedule\n- ","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Evak","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation\nTODO: Take off last layer, get just logits ","metadata":{}},{"cell_type":"code","source":"\nimport json\n\nwith open('/kaggle/input/birdclef-2021-mel-spectograms/LABEL_IDS.json') as f:\n    class_map = json.load(f) \n    \ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nNUM_CLASSES = 397\n\n\n# Reverse class map for index → bird label\nidx_to_bird = {v: k for k, v in class_map.items()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:25:38.786984Z","iopub.execute_input":"2025-04-21T10:25:38.787276Z","iopub.status.idle":"2025-04-21T10:25:38.802939Z","shell.execute_reply.started":"2025-04-21T10:25:38.787254Z","shell.execute_reply":"2025-04-21T10:25:38.802216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\n# Load trained model\nmodel = EfficientNet.from_name('efficientnet-b0', num_classes=NUM_CLASSES)\nmodel.load_state_dict(torch.load('/kaggle/input/iml-efficientnet/pytorch/softmax_ce/1/best_model_weights.pth', map_location=device))\nmodel.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:26:26.214744Z","iopub.execute_input":"2025-04-21T10:26:26.215437Z","iopub.status.idle":"2025-04-21T10:26:26.371413Z","shell.execute_reply.started":"2025-04-21T10:26:26.21541Z","shell.execute_reply":"2025-04-21T10:26:26.370829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from glob import glob \nfrom tqdm.notebook import tqdm  # Import tqdm for notebooks\nimport torch.nn.functional as F\n\nTHRESHOLD = 0.2\nTOP_K = 3\n\n# Inference loop\ntest_dir = '/kaggle/input/birdclef-2021-mel-spectograms/test_mel_spectograms/test_mel_spectograms'\ntest_files = glob(os.path.join(test_dir, '*.npy'))\n\nprint(os.listdir(test_dir))\nprint(test_files)\nrows = []\n\n# Transformation to match training input\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n])\n\nfor file_path in tqdm(test_files):\n    audio_id = os.path.basename(file_path).replace('.npy', '')\n    spec_stack = np.load(file_path)  # shape: [120, 128, 321]\n\n    for i in range(spec_stack.shape[0]):\n        spec = spec_stack[i]  # [128, 321]\n        spec_tensor = torch.tensor(spec).float().unsqueeze(0)  # [1, 128, 321]\n        spec_tensor = spec_tensor.repeat(3, 1, 1)  # [3, 128, 321]\n        spec_tensor = transform(spec_tensor).unsqueeze(0).to(device)  # [1, 3, 224, 224]\n\n        with torch.no_grad():\n            logits = model(spec_tensor)\n            probs = F.softmax(logits, dim=1).squeeze()  # [NUM_CLASSES]\n\n        top_probs, top_idxs = torch.topk(probs, TOP_K)\n        labels = [idx_to_bird[idx.item()] for idx, prob in zip(top_idxs, top_probs) if prob.item() > THRESHOLD]\n\n        if not labels:\n            labels = ['nocall']\n\n        row_id = f\"{audio_id}_{(i+1)*5}\"\n        rows.append({\n            'row_id': row_id,\n            'birds': ' '.join(labels)\n        })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:35:35.864971Z","iopub.execute_input":"2025-04-21T10:35:35.865654Z","iopub.status.idle":"2025-04-21T10:36:12.818638Z","shell.execute_reply.started":"2025-04-21T10:35:35.865627Z","shell.execute_reply":"2025-04-21T10:36:12.817845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Final predictions DataFrame\neval_df = pd.DataFrame(rows)\neval_df[eval_df['birds']!='nocall'].head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-21T10:36:35.905624Z","iopub.execute_input":"2025-04-21T10:36:35.906338Z","iopub.status.idle":"2025-04-21T10:36:35.916627Z","shell.execute_reply.started":"2025-04-21T10:36:35.906311Z","shell.execute_reply":"2025-04-21T10:36:35.915847Z"}},"outputs":[],"execution_count":null}]}