{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"***1. Введение***\n\n Целью данного проекта является разработка и создание модели классификации голосов птиц на основе набора данных BirdCLEF 2021. Основная задача состоит в том, чтобы создать нейронную сеть, способную идентифицировать вид птицы по аудиоклипам и достичь высокой точности валидации ($\\text{Validation Accuracy}$) ($\\geq 60\\%$).","metadata":{}},{"cell_type":"markdown","source":"***2. Методология Проекта***\n\nДля достижения высокой точности в классификации голосов птиц был реализован структурированный подход, начавшийся с подготовки данных, где использовался полный набор аудиофайлов BirdCLEF 2021, а дисбаланс классов был устранен с помощью механизма $\\text{WeightedRandomSampler}$, чтобы гарантировать адекватное представление редких видов. Затем, на этапе извлечения признаков, каждый 5-секундный клип был преобразован в Mel-спектрограмму  с последующей нормализацией, что сделало аудиоданные пригодными для обработки CNN. В качестве архитектуры модели использовался метод $\\text{Transfer Learning}$ на основе мощных $\\text{backbone}$ (таких как $\\text{EfficientNet}$ или $\\text{ResNet}$) из библиотеки $\\text{timm}$, к которым был добавлен специализированный классификационный $\\text{head}$. Наконец, для оптимизации и регуляризации применялся комплекс передовых техник: $\\text{Mixup}$ и $\\text{SpecAugment}$ для аугментации данных, $\\text{Label Smoothing}$ для повышения устойчивости, $\\text{Learning Rate Scheduler}$ (с $\\text{Warm-up}$ и $\\text{Cosine Decay}$) для стабильного обучения, а также $\\text{Early Stopping}$ для предотвращения переобучения ($\\text{Overfitting}$).","metadata":{}},{"cell_type":"code","source":"# run this cell once\n!pip install timm --quiet\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:42:51.555701Z","iopub.execute_input":"2025-11-17T08:42:51.556037Z","iopub.status.idle":"2025-11-17T08:42:54.882802Z","shell.execute_reply.started":"2025-11-17T08:42:51.556012Z","shell.execute_reply":"2025-11-17T08:42:54.881776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 1: imports and device\nimport os\nimport math\nimport random\nimport time\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\n\nimport librosa\nimport timm\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"Device:\", DEVICE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:03.291281Z","iopub.execute_input":"2025-11-17T08:43:03.292121Z","iopub.status.idle":"2025-11-17T08:43:06.478649Z","shell.execute_reply.started":"2025-11-17T08:43:03.292088Z","shell.execute_reply":"2025-11-17T08:43:06.47795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 2: collect ALL audio files from train_short_audio\nDATA_DIR = \"/kaggle/input/birdclef-2021/train_short_audio\"\nassert os.path.isdir(DATA_DIR), f\"{DATA_DIR} not found\"\n\nfilepaths = []\nlabels = []\nlabel2idx = {}\n\nfor idx, bird in enumerate(sorted(os.listdir(DATA_DIR))):\n    bird_dir = os.path.join(DATA_DIR, bird)\n    if not os.path.isdir(bird_dir):\n        continue\n    label2idx[bird] = idx\n    # collect .ogg and .wav\n    for f in sorted(os.listdir(bird_dir)):\n        if f.endswith(\".ogg\") or f.endswith(\".wav\"):\n            filepaths.append(os.path.join(bird_dir, f))\n            labels.append(idx)\n\ndf = pd.DataFrame({\"filepath\": filepaths, \"label_idx\": labels})\nprint(\"Total files:\", len(df), \"Classes:\", len(label2idx))\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:14.225674Z","iopub.execute_input":"2025-11-17T08:43:14.226043Z","iopub.status.idle":"2025-11-17T08:43:14.86049Z","shell.execute_reply.started":"2025-11-17T08:43:14.226017Z","shell.execute_reply":"2025-11-17T08:43:14.85976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 3: split into train/valid stratified by label\nfrom sklearn.model_selection import train_test_split\ntrain_df, valid_df = train_test_split(df, test_size=0.1, stratify=df.label_idx, random_state=42)\ntrain_df = train_df.reset_index(drop=True)\nvalid_df = valid_df.reset_index(drop=True)\nprint(\"Train:\", len(train_df), \"Valid:\", len(valid_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:17.665765Z","iopub.execute_input":"2025-11-17T08:43:17.666398Z","iopub.status.idle":"2025-11-17T08:43:18.27992Z","shell.execute_reply.started":"2025-11-17T08:43:17.666371Z","shell.execute_reply":"2025-11-17T08:43:18.279126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 4: Precompute mel spectrograms and save to disk\nMEL_DIR = \"/kaggle/working/mel_cache_v2\"\nos.makedirs(MEL_DIR, exist_ok=True)\n\nSR = 32000\nDURATION = 5      # seconds\nSAMPLES = SR * DURATION\nN_MELS = 128\nN_FFT = 1024\nHOP = 320\nFMIN = 50\nFMAX = 14000\n\ndef compute_and_save_mel(filepath):\n    base = os.path.basename(filepath)\n    out_path = os.path.join(MEL_DIR, base + \".npy\")\n    if os.path.exists(out_path):\n        return out_path\n    try:\n        y, sr = librosa.load(filepath, sr=SR, mono=True, duration=DURATION)\n        if len(y) < SAMPLES:\n            y = np.pad(y, (0, SAMPLES - len(y)))\n        else:\n            y = y[:SAMPLES]\n        mel = librosa.feature.melspectrogram(y=y, sr=SR, n_mels=N_MELS,\n                                             n_fft=N_FFT, hop_length=HOP, fmin=FMIN, fmax=FMAX)\n        mel_db = librosa.power_to_db(mel, ref=np.max)\n        np.save(out_path, mel_db.astype(np.float32))\n        return out_path\n    except Exception as e:\n        print(\"Error computing mel for\", filepath, e)\n        return None\n\n# Precompute for train and valid (progress bar)\nprint(\"Precomputing train mels...\")\nfor p in tqdm(train_df.filepath.values, total=len(train_df)):\n    compute_and_save_mel(p)\n\nprint(\"Precomputing valid mels...\")\nfor p in tqdm(valid_df.filepath.values, total=len(valid_df)):\n    compute_and_save_mel(p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:20.697406Z","iopub.execute_input":"2025-11-17T08:43:20.698295Z","iopub.status.idle":"2025-11-17T08:43:21.06709Z","shell.execute_reply.started":"2025-11-17T08:43:20.69827Z","shell.execute_reply":"2025-11-17T08:43:21.066396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 5: Dataset (loads precomputed mel .npy)\nclass MelDataset(Dataset):\n    def __init__(self, df, mel_dir=MEL_DIR, n_mels=N_MELS, augment=False):\n        self.df = df.reset_index(drop=True)\n        self.mel_dir = mel_dir\n        self.n_mels = n_mels\n        self.augment = augment\n\n    def __len__(self):\n        return len(self.df)\n\n    def spec_augment(self, x, time_mask_max=40, freq_mask_max=12):\n        # simple SpecAugment inplace on numpy\n        x = x.copy()\n        T = x.shape[1]\n        F = x.shape[0]\n        # freq mask\n        f = np.random.randint(0, freq_mask_max)\n        f0 = np.random.randint(0, F - f + 1) if F - f + 1 > 0 else 0\n        x[f0:f0+f, :] = 0\n        # time mask\n        t = np.random.randint(0, time_mask_max)\n        t0 = np.random.randint(0, T - t + 1) if T - t + 1 > 0 else 0\n        x[:, t0:t0+t] = 0\n        return x\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        base = os.path.basename(row.filepath)\n        mel_path = os.path.join(self.mel_dir, base + \".npy\")\n        mel = np.load(mel_path)  # shape: (n_mels, time)\n        # augmentation (specaugment) on train only\n        if self.augment:\n            mel = self.spec_augment(mel)\n        # compute delta and delta-delta\n        delta = librosa.feature.delta(mel)\n        delta2 = librosa.feature.delta(mel, order=2)\n        # stack to 3 channels\n        arr = np.stack([mel, delta, delta2], axis=0).astype(np.float32)\n        # normalize per-sample (mean/std)\n        arr = (arr - arr.mean()) / (arr.std() + 1e-6)\n        return torch.from_numpy(arr), int(row.label_idx)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:29.984486Z","iopub.execute_input":"2025-11-17T08:43:29.985275Z","iopub.status.idle":"2025-11-17T08:43:29.992804Z","shell.execute_reply.started":"2025-11-17T08:43:29.985248Z","shell.execute_reply":"2025-11-17T08:43:29.991993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6: prepare balanced sampler and dataloaders (num_workers=0 for stability on Kaggle)\ntrain_ds = MelDataset(train_df, augment=True)\nvalid_ds = MelDataset(valid_df, augment=False)\n\n# WeightedRandomSampler for class balance\ncounts = train_df.label_idx.value_counts().sort_index().values\nclass_weights = 1.0 / (counts + 1e-12)\nsample_weights = class_weights[train_df.label_idx.values]\nsampler = WeightedRandomSampler(weights=sample_weights, num_samples=len(sample_weights), replacement=True)\n\nBATCH_SIZE = 32  # try 32, reduce if OOM\ntrain_dl = DataLoader(train_ds, batch_size=BATCH_SIZE, sampler=sampler, num_workers=0, pin_memory=True)\nvalid_dl = DataLoader(valid_ds, batch_size=BATCH_SIZE, shuffle=False, num_workers=0, pin_memory=True)\n\nprint(\"Train batches:\", len(train_dl), \"Valid batches:\", len(valid_dl))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:34.138534Z","iopub.execute_input":"2025-11-17T08:43:34.139056Z","iopub.status.idle":"2025-11-17T08:43:34.151808Z","shell.execute_reply.started":"2025-11-17T08:43:34.13903Z","shell.execute_reply":"2025-11-17T08:43:34.150851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 7: Build classifier using timm backbone (pretrained on ImageNet)\n# We will use an image model and adapt first conv to accept 3-channel mel input (already 3-ch)\nbackbone_name = \"tf_efficientnet_b3_ns\"  # good balance of accuracy/speed; change if you want larger model\nmodel = timm.create_model(backbone_name, pretrained=True, num_classes=0)  # num_classes=0 -> feature extractor\nfeat_dim = model.num_features\nprint(\"Backbone feature dim:\", feat_dim)\n\n# classifier head\nclass AudioClassifier(nn.Module):\n    def __init__(self, backbone, feat_dim, n_classes, dropout=0.5):\n        super().__init__()\n        self.backbone = backbone\n        self.pool = nn.AdaptiveAvgPool2d((1,1))\n        self.fc1 = nn.Linear(feat_dim, 512)\n        self.dropout = nn.Dropout(dropout)\n        self.fc2 = nn.Linear(512, n_classes)\n\n    def forward(self, x):\n        # x: B x 3 x H x W  (we currently have mel shaped as [3, n_mels, time])\n        feats = self.backbone.forward_features(x)  # backbone-specific API\n        # if backbone returns 4D feature map, pool it\n        if feats.ndim == 4:\n            feats = self.pool(feats).view(feats.size(0), -1)\n        out = F.relu(self.fc1(feats))\n        out = self.dropout(out)\n        out = self.fc2(out)\n        return out\n\nn_classes = len(label2idx)\nmodel = AudioClassifier(model, feat_dim, n_classes).to(DEVICE)\nprint(model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:36.697044Z","iopub.execute_input":"2025-11-17T08:43:36.697328Z","iopub.status.idle":"2025-11-17T08:43:37.559802Z","shell.execute_reply.started":"2025-11-17T08:43:36.69731Z","shell.execute_reply":"2025-11-17T08:43:37.558593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 8: mixup, label smoothing support\ndef mixup_data(x, y, alpha=0.4):\n    if alpha <= 0:\n        return x, y, None, None, 1.0\n    lam = np.random.beta(alpha, alpha)\n    batch_size = x.size()[0]\n    index = torch.randperm(batch_size).to(DEVICE)\n    mixed_x = lam * x + (1 - lam) * x[index, :]\n    y_a, y_b = y, y[index]\n    return mixed_x, y_a, y_b, index, lam\n\n# CrossEntropy with label smoothing via built-in parameter (if supported), else manual\ncriterion = nn.CrossEntropyLoss(label_smoothing=0.1)  # requires PyTorch >=1.10\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:45.794059Z","iopub.execute_input":"2025-11-17T08:43:45.794354Z","iopub.status.idle":"2025-11-17T08:43:45.799515Z","shell.execute_reply.started":"2025-11-17T08:43:45.794334Z","shell.execute_reply":"2025-11-17T08:43:45.798697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 9: optimizer and LR schedule (AdamW + warmup + cosine)\nfrom torch.optim import AdamW\n\nLR = 1e-3  # relatively high initial LR\nWEIGHT_DECAY = 1e-4\nEPOCHS = 80\n\noptimizer = AdamW(model.parameters(), lr=LR, weight_decay=WEIGHT_DECAY)\n\n# Warmup + CosineAnnealing\ndef get_scheduler(optimizer, warmup_epochs, total_epochs):\n    def lr_lambda(epoch):\n        if epoch < warmup_epochs:\n            return float(epoch) / float(max(1, warmup_epochs))\n        # cosine decay after warmup\n        progress = float(epoch - warmup_epochs) / float(max(1, total_epochs - warmup_epochs))\n        return 0.5 * (1.0 + math.cos(math.pi * progress))\n    return torch.optim.lr_scheduler.LambdaLR(optimizer, lr_lambda)\n\nwarmup_epochs = 5\nscheduler = get_scheduler(optimizer, warmup_epochs=warmup_epochs, total_epochs=EPOCHS)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:48.368666Z","iopub.execute_input":"2025-11-17T08:43:48.369253Z","iopub.status.idle":"2025-11-17T08:43:48.376415Z","shell.execute_reply.started":"2025-11-17T08:43:48.369226Z","shell.execute_reply":"2025-11-17T08:43:48.375583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 10: training loop with mixup and early stopping\nbest_val = 0.0\npatience = 8\nno_improve = 0\nsave_path = \"best_model_resnet_audio.pth\"\n\nfor epoch in range(1, EPOCHS+1):\n    model.train()\n    train_loss = 0.0\n    correct = 0\n    total = 0\n\n    for xb, yb in tqdm(train_dl, desc=f\"Epoch {epoch} train\"):\n        xb = xb.to(DEVICE)            # shape B x 3 x n_mels x time\n        yb = yb.to(DEVICE).long()\n\n        # optionally resize/exact shape for image backbone:\n        # timm backbones expect HxW ~ 224x224-ish. We need to resize mel to that.\n        # Do it here using torch.nn.functional.interpolate\n        xb = F.interpolate(xb, size=(224, 224), mode='bilinear', align_corners=False)\n\n        # mixup\n        xb, y_a, y_b, index, lam = mixup_data(xb, yb, alpha=0.4)\n        optimizer.zero_grad()\n        outputs = model(xb)\n        # mixup loss\n        loss = lam * criterion(outputs, y_a) + (1 - lam) * criterion(outputs, y_b)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item() * xb.size(0)\n        preds = outputs.argmax(dim=1)\n        # for accuracy approximate using y_a\n        correct += (preds == yb).sum().item()\n        total += xb.size(0)\n\n    scheduler.step()\n    train_loss = train_loss / total\n    train_acc = correct / total\n\n    # validation\n    model.eval()\n    val_loss = 0.0\n    val_correct = 0\n    val_total = 0\n    with torch.no_grad():\n        for xb, yb in tqdm(valid_dl, desc=f\"Epoch {epoch} valid\"):\n            xb = xb.to(DEVICE)\n            yb = yb.to(DEVICE).long()\n            xb = F.interpolate(xb, size=(224,224), mode='bilinear', align_corners=False)\n            outputs = model(xb)\n            loss = criterion(outputs, yb)\n            val_loss += loss.item() * xb.size(0)\n            preds = outputs.argmax(dim=1)\n            val_correct += (preds == yb).sum().item()\n            val_total += xb.size(0)\n\n    val_loss = val_loss / val_total\n    val_acc = val_correct / val_total\n\n    print(f\"Epoch {epoch} Train Loss {train_loss:.4f} Train Acc {train_acc:.4f} | Val Loss {val_loss:.4f} Val Acc {val_acc:.4f}\")\n\n    # save best\n    if val_acc > best_val:\n        best_val = val_acc\n        torch.save({\n            'epoch': epoch,\n            'model_state': model.state_dict(),\n            'optimizer_state': optimizer.state_dict(),\n            'best_val': best_val,\n            'label2idx': label2idx\n        }, save_path)\n        print(\"Saved best model:\", save_path)\n        no_improve = 0\n    else:\n        no_improve += 1\n        if no_improve >= patience:\n            print(\"Early stopping triggered.\")\n            break\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-17T08:43:51.827413Z","iopub.execute_input":"2025-11-17T08:43:51.828018Z","execution_failed":"2025-11-17T19:50:33.349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**3. Результаты Проекта**\n\nПосле 62 эпох обучения, разработанная модель достигла впечатляющего результата: точность валидации ($\\text{Val Acc}$) составила $\\text{73.41\\%}$ ($\\text{Val Acc} = 0.7341$). Этот показатель значительно превысил целевой порог в $\\text{60\\%}$, что однозначно подтверждает, что модель эффективно освоила сложные акустические паттерны из набора данных BirdCLEF. Такой высокий уровень точности является прямым следствием успешного применения комплекса продвинутых методик, включая балансировку данных ($\\text{WeightedRandomSampler}$), современные аугментации ($\\text{Mixup}$, $\\text{SpecAugment}$) и использование мощной архитектуры $\\text{backbone}$, что в итоге позволило создать высокоэффективный классификатор для автоматической идентификации видов птиц по их вокализациям.","metadata":{}}]}