{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":14176592,"sourceType":"datasetVersion","datasetId":9036810}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Подготовка данных**","metadata":{}},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport librosa\nfrom tqdm import tqdm\nimport math\nimport json\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\nimport torchaudio\nfrom torchvision.models import efficientnet_b0, EfficientNet_B0_Weights\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:30.422909Z","iopub.execute_input":"2025-12-17T07:54:30.423126Z","iopub.status.idle":"2025-12-17T07:54:40.630067Z","shell.execute_reply.started":"2025-12-17T07:54:30.423104Z","shell.execute_reply":"2025-12-17T07:54:40.629506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_ROOT = Path(\"/kaggle/input/birdclef-2021\")\nOUT_DIR = Path(\"./output\")\nCACHE_DIR = OUT_DIR / \"mel_cache\"\nCHECKPOINT_DIR = OUT_DIR / \"checkpoints\"\nOUT_DIR.mkdir(exist_ok=True)\nCACHE_DIR.mkdir(exist_ok=True)\nCHECKPOINT_DIR.mkdir(exist_ok=True)\nWEIGHTS_PATH = (\n    \"/kaggle/input/efficientnet-b0-imagenet-weights/\"\n    \"efficientnet_b0_imagenet.pth\"\n)\n\n\nSAMPLE_RATE = 32000\nN_MELS = 128\nCLIP_DURATION = 10\nCLIP_SAMPLES = SAMPLE_RATE * CLIP_DURATION\nHOP_LENGTH = 512\nN_FFT = 2048\n\nBATCH_SIZE = 32\nEPOCHS = 8\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nSEED = 42","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.631756Z","iopub.execute_input":"2025-12-17T07:54:40.632119Z","iopub.status.idle":"2025-12-17T07:54:40.638609Z","shell.execute_reply.started":"2025-12-17T07:54:40.632101Z","shell.execute_reply":"2025-12-17T07:54:40.637147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_meta = pd.read_csv(DATA_ROOT / \"train_metadata.csv\")\nprint(\"Train metadata:\", train_meta.shape)\nspecies = sorted(train_meta[\"primary_label\"].unique())\nlabel2idx = {lbl: i for i, lbl in enumerate(species)}\nidx2label = {i: lbl for lbl, i in label2idx.items()}\nNUM_CLASSES = len(species)\nprint(\"Num species:\", NUM_CLASSES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.639902Z","iopub.status.idle":"2025-12-17T07:54:40.640127Z","shell.execute_reply.started":"2025-12-17T07:54:40.640024Z","shell.execute_reply":"2025-12-17T07:54:40.640034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def wav_to_log_mel(wave: np.ndarray, sr=SAMPLE_RATE, n_mels=N_MELS, n_fft=N_FFT, hop_length=HOP_LENGTH):\n    mel = librosa.feature.melspectrogram(y=wave, sr=sr, n_fft=n_fft, hop_length=hop_length, n_mels=n_mels, power=2.0)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n    mel_db = (mel_db + 80.0) / 80.0\n    return mel_db.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.640917Z","iopub.status.idle":"2025-12-17T07:54:40.641200Z","shell.execute_reply.started":"2025-12-17T07:54:40.641041Z","shell.execute_reply":"2025-12-17T07:54:40.641054Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Предобработка**","metadata":{}},{"cell_type":"code","source":"def cache_mel_for_file(filepath: Path, out_path: Path):\n    y, sr = librosa.load(filepath, sr=SAMPLE_RATE, mono=True)\n    total_samples = len(y)\n    n_segments = max(1, total_samples // CLIP_SAMPLES)\n    saved = []\n    for i in range(n_segments):\n        start = i * CLIP_SAMPLES\n        clip = y[start:start + CLIP_SAMPLES]\n        if len(clip) < CLIP_SAMPLES:\n            clip = np.pad(clip, (0, CLIP_SAMPLES - len(clip)))\n        mel = wav_to_log_mel(clip)\n        seg_path = out_path / f\"{filepath.stem}_seg{i}.npy\"\n        if not seg_path.exists():\n            np.save(seg_path, mel)\n        saved.append(seg_path)\n    return saved\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.642657Z","iopub.status.idle":"2025-12-17T07:54:40.642922Z","shell.execute_reply.started":"2025-12-17T07:54:40.642801Z","shell.execute_reply":"2025-12-17T07:54:40.642815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ShortAudioDataset(Dataset):\n    def __init__(self, df_meta, data_root, label2idx,\n                 cache_dir=CACHE_DIR, training=True):\n        self.df = df_meta.reset_index(drop=True)\n        self.data_root = Path(data_root) / \"train_short_audio\"\n        self.label2idx = label2idx\n        self.training = training\n        self.cache_dir = Path(cache_dir)\n        self.cache_dir.mkdir(parents=True, exist_ok=True)\n\n        self.freq_mask = torchaudio.transforms.FrequencyMasking(freq_mask_param=16)\n        self.time_mask = torchaudio.transforms.TimeMasking(time_mask_param=24)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        species = row[\"primary_label\"]\n        filename = row[\"filename\"]\n\n        audio_path = self.data_root / species / filename\n        if not audio_path.exists():\n            found = list(self.data_root.rglob(filename))\n            if found:\n                audio_path = found[0]\n            else:\n                raise FileNotFoundError(audio_path)\n\n        mel_cache = self.cache_dir / f\"{audio_path.stem}.npy\"\n\n        if not mel_cache.exists():\n            y, _ = librosa.load(audio_path, sr=SAMPLE_RATE, mono=True)\n\n            if len(y) < CLIP_SAMPLES:\n                y = np.pad(y, (0, CLIP_SAMPLES - len(y)))\n            else:\n                if self.training:\n                    start = np.random.randint(0, len(y) - CLIP_SAMPLES)\n                    y = y[start:start + CLIP_SAMPLES]\n                else:\n                    y = y[:CLIP_SAMPLES]\n\n            mel = wav_to_log_mel(y)\n            np.save(mel_cache, mel)\n        else:\n            mel = np.load(mel_cache)\n\n        mel = torch.from_numpy(mel).unsqueeze(0)  # (1, n_mels, T)\n\n        if self.training:\n            if torch.rand(1).item() < 0.5:\n                mel = self.freq_mask(mel)\n            if torch.rand(1).item() < 0.5:\n                mel = self.time_mask(mel)\n\n        label = self.label2idx[row[\"primary_label\"]]\n        return mel, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.643634Z","iopub.status.idle":"2025-12-17T07:54:40.643836Z","shell.execute_reply.started":"2025-12-17T07:54:40.643739Z","shell.execute_reply":"2025-12-17T07:54:40.643748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import timm\nimport torch\nimport torch.nn as nn\nfrom torchvision.models import efficientnet_b0\n\nWEIGHTS_PATH = (\n    \"/kaggle/input/efficientnet-b0-imagenet-weights/efficientnet_b0_imagenet.pth\"\n)\n\nclass EfficientNetWrapper(nn.Module):\n    def __init__(self, out_dim=512):\n        super().__init__()\n\n        self.net = efficientnet_b0(weights=None)\n\n        state_dict = torch.load(WEIGHTS_PATH, map_location=\"cpu\")\n        self.net.load_state_dict(state_dict, strict=False)\n\n        old_conv = self.net.features[0][0]\n        self.net.features[0][0] = nn.Conv2d(\n            in_channels=1,\n            out_channels=old_conv.out_channels,\n            kernel_size=old_conv.kernel_size,\n            stride=old_conv.stride,\n            padding=old_conv.padding,\n            bias=False,\n        )\n\n        self.net.classifier = nn.Identity()\n\n        self.head = nn.Sequential(\n            nn.Linear(1280, out_dim),\n            nn.BatchNorm1d(out_dim),\n        )\n\n    def forward(self, x):\n        x = self.net(x)\n        x = self.head(x)\n        return x\n\n\n\nUSE_TIMM = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.644437Z","iopub.status.idle":"2025-12-17T07:54:40.644716Z","shell.execute_reply.started":"2025-12-17T07:54:40.644585Z","shell.execute_reply":"2025-12-17T07:54:40.644596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdClassifier(nn.Module):\n    def __init__(self, backbone, out_dim=512, n_classes=NUM_CLASSES):\n        super().__init__()\n        self.backbone = backbone\n        self.proj = nn.Linear(out_dim, n_classes)\n\n    def forward(self, x):\n        feats = self.backbone(x)  # (B, out_dim)\n        logits = self.proj(feats)\n        return logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.645755Z","iopub.status.idle":"2025-12-17T07:54:40.646038Z","shell.execute_reply.started":"2025-12-17T07:54:40.645910Z","shell.execute_reply":"2025-12-17T07:54:40.645927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train_meta, test_size=0.2, random_state=SEED, stratify=train_meta[\"primary_label\"])\n\ntrain_ds = ShortAudioDataset(train_df, DATA_ROOT, label2idx, training=True)\nvalid_ds = ShortAudioDataset(valid_df, DATA_ROOT, label2idx, training=True)\n\ntrain_loader = DataLoader(train_ds, batch_size=BATCH_SIZE, shuffle=True, num_workers=4, pin_memory=True)\nvalid_loader = DataLoader(valid_ds, batch_size=BATCH_SIZE, shuffle=False, num_workers=4, pin_memory=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.647589Z","iopub.status.idle":"2025-12-17T07:54:40.647903Z","shell.execute_reply.started":"2025-12-17T07:54:40.647743Z","shell.execute_reply":"2025-12-17T07:54:40.647757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"backbone = EfficientNetWrapper(out_dim=512)\nmodel = BirdClassifier(\n    backbone,\n    out_dim=512,\n    n_classes=NUM_CLASSES\n).to(DEVICE)\n\nfor p in model.backbone.parameters():\n    p.requires_grad = False\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, factor=0.5, patience=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.648931Z","iopub.status.idle":"2025-12-17T07:54:40.650206Z","shell.execute_reply.started":"2025-12-17T07:54:40.650091Z","shell.execute_reply":"2025-12-17T07:54:40.650104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_f1 = 0.0\nfor epoch in range(1, EPOCHS + 1):\n    if (epoch == 3):\n        for p in model.backbone.parameters():\n            p.requires_grad = True\n\n    model.train()\n    losses = []\n    for x, y in tqdm(train_loader, desc=f\"Train {epoch}\"):\n        x = x.to(DEVICE)\n        y = y.to(DEVICE)\n        optimizer.zero_grad()\n        logits = model(x)\n        loss = criterion(logits, y)\n        loss.backward()\n        optimizer.step()\n        losses.append(loss.item())\n    avg_loss = np.mean(losses)\n\n    # валидация\n    model.eval()\n    preds = []\n    targs = []\n    with torch.no_grad():\n        for x, y in tqdm(valid_loader, desc=f\"Val {epoch}\"):\n            x = x.to(DEVICE)\n            y = y.to(DEVICE)\n            logits = model(x)\n            pred = logits.argmax(dim=1)\n            preds.extend(pred.cpu().numpy())\n            targs.extend(y.cpu().numpy())\n    val_f1 = f1_score(targs, preds, average='macro')\n    print(f\"Epoch {epoch}: train_loss={avg_loss:.4f} val_f1={val_f1:.4f}\")\n    scheduler.step(val_f1)\n    if val_f1 > best_f1:\n        best_f1 = val_f1\n        torch.save(model.state_dict(), CHECKPOINT_DIR / f\"best_epoch{epoch}_f1{val_f1:.4f}.pt\")\nprint(\"Best val f1:\", best_f1)\n\nTEST_SSG_DIR = DATA_ROOT / \"test_soundscapes\"\n\nTOP_K = 3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.651008Z","iopub.status.idle":"2025-12-17T07:54:40.651211Z","shell.execute_reply.started":"2025-12-17T07:54:40.651113Z","shell.execute_reply":"2025-12-17T07:54:40.651122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def infer_soundscape_file(model, filepath: Path, top_k=TOP_K):\n    # возвращает список (end_time_sec, [label_codes...])\n    y, sr = librosa.load(filepath, sr=SAMPLE_RATE, mono=True)\n    total_seconds = math.ceil(len(y) / SAMPLE_RATE)\n    results = []\n    n_segments = int(len(y) // CLIP_SAMPLES)\n    for i in range(n_segments):\n        start = i * CLIP_SAMPLES\n        clip = y[start:start + CLIP_SAMPLES]\n        if len(clip) < CLIP_SAMPLES:\n            clip = np.pad(clip, (0, CLIP_SAMPLES - len(clip)))\n        mel = wav_to_log_mel(clip)          # (n_mels, T)\n        x = torch.from_numpy(mel).unsqueeze(0).unsqueeze(0).to(DEVICE)  # (1,1,n_mels,T)\n        with torch.no_grad():\n            logits = model(x)\n            probs = torch.softmax(logits, dim=1).cpu().numpy()[0]\n        top_idx = probs.argsort()[-top_k:][::-1]\n        top_labels = [idx2label[i] for i in top_idx]\n        end_time = (i+1) * CLIP_DURATION   # конец окна в секундах\n        results.append((end_time, top_labels))\n    return results\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.652322Z","iopub.status.idle":"2025-12-17T07:54:40.652611Z","shell.execute_reply.started":"2025-12-17T07:54:40.652480Z","shell.execute_reply":"2025-12-17T07:54:40.652494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_rows = []\nfor filepath in tqdm(sorted(TEST_SSG_DIR.glob(\"*.ogg\")), desc=\"Soundscapes\"):\n    soundscape_id = filepath.stem  # обычно формат like \"12345_SSW_20170429\" или id — для row_id обычно нужен только часть перед расширением.\n    preds = infer_soundscape_file(model, filepath, top_k=TOP_K)\n    for end_time, labels in preds:\n        # row_id по формату соревнования: soundscape_[soundscape_id]_[end_time]\n        row_id = f\"soundscape_{soundscape_id}_{end_time}\"\n        # birds — пробел разделённые коды (или 'nocall' если пусто)\n        birds = \" \".join(labels) if labels else \"nocall\"\n        submission_rows.append({\"row_id\": row_id, \"birds\": birds})\n\nsubmission_df = pd.DataFrame(submission_rows)\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(\"Saved submission:\", \"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T07:54:40.653861Z","iopub.status.idle":"2025-12-17T07:54:40.654117Z","shell.execute_reply.started":"2025-12-17T07:54:40.653990Z","shell.execute_reply":"2025-12-17T07:54:40.654003Z"}},"outputs":[],"execution_count":null}]}