{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Классификация звуков птиц — BirdCLEF 2021","metadata":{}},{"cell_type":"markdown","source":"# 1. Постановка задачи\n\n1. BirdCLEF 2021: по длинным аудиозаписям soundscapes нужно для каждого 5-секундного окна предсказать множество видов птиц, которые “звучат” в этом окне. В одном окне может быть несколько видов, либо ни одного (тогда метка nocall). \n2. Для каждой строки row_id требуется поле birds — строка со списком видов через пробел либо nocall. \n3. оревнование оценивается по row-wise micro averaged F1 (F1 считается по каждой строке/окну на множествах предсказанных и истинных видов).","metadata":{}},{"cell_type":"code","source":"import os, re, math\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\nimport librosa as lb\nimport torch\nfrom torch import nn","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:01:24.837649Z","iopub.execute_input":"2025-12-27T00:01:24.837920Z","iopub.status.idle":"2025-12-27T00:01:29.750275Z","shell.execute_reply.started":"2025-12-27T00:01:24.837886Z","shell.execute_reply":"2025-12-27T00:01:29.749533Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 0) Config","metadata":{}},{"cell_type":"code","source":"class CFG:\n    SR = 32000\n    SEG_SECONDS = 5\n    SEG_SAMPLES = SR * SEG_SECONDS\n\n    N_MELS = 128\n    FMIN = 0\n    FMAX = 16000\n\n    # (как в популярных публичных ноутбуках по BirdCLEF 2021)\n    N_FFT = SR // 10\n    HOP = SR // 40\n\n    THRESH = 0.25\n    BATCH = 16\n\n    DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nDATA = Path(\"/kaggle/input/birdclef-2021\")\nTEST_ROOT = DATA / \"test_soundscapes\"\nSAMPLE_SUB = DATA / \"sample_submission.csv\"\nTARGET = DATA / \"train_soundscape_labels.csv\"\n\n# offline fallback\nif not list(TEST_ROOT.glob(\"*.ogg\")):\n    TEST_ROOT = DATA / \"train_soundscapes\"\n    SAMPLE_SUB = None\n\nprint(\"DEVICE:\", CFG.DEVICE)\nprint(\"AUDIO ROOT:\", TEST_ROOT)\nprint(\"SAMPLE_SUB:\", SAMPLE_SUB)\nprint(\"TARGET:\", TARGET if TARGET.exists() else None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:01:29.751703Z","iopub.execute_input":"2025-12-27T00:01:29.752109Z","iopub.status.idle":"2025-12-27T00:01:29.915837Z","shell.execute_reply.started":"2025-12-27T00:01:29.752074Z","shell.execute_reply":"2025-12-27T00:01:29.915066Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1) Labels","metadata":{}},{"cell_type":"code","source":"meta = pd.read_csv(DATA / \"train_metadata.csv\")\nLABELS = sorted(meta[\"primary_label\"].unique())\nLABEL2ID = {l:i for i,l in enumerate(LABELS)}\nID2LABEL = {i:l for l,i in LABEL2ID.items()}\nNUM_CLASSES = len(LABELS)\nprint(\"NUM_CLASSES:\", NUM_CLASSES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:01:29.916673Z","iopub.execute_input":"2025-12-27T00:01:29.916876Z","iopub.status.idle":"2025-12-27T00:01:30.614003Z","shell.execute_reply.started":"2025-12-27T00:01:29.916857Z","shell.execute_reply":"2025-12-27T00:01:30.613216Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2) (Optional) site prior from train_soundscape_labels.csv","metadata":{}},{"cell_type":"code","source":"SITE_SPECIES = {}\nif TARGET.exists():\n    df_site = pd.read_csv(TARGET, usecols=[\"site\", \"birds\"])\n    for site, birds in zip(df_site[\"site\"], df_site[\"birds\"]):\n        if isinstance(birds, str) and birds != \"nocall\":\n            SITE_SPECIES.setdefault(site, set()).update(birds.split())\n\n# convert site prior -> index masks (fast)\nSITE_MASK = {}\nfor site, sp in SITE_SPECIES.items():\n    idx = [LABEL2ID[s] for s in sp if s in LABEL2ID]\n    if idx:\n        m = torch.zeros(NUM_CLASSES, dtype=torch.float32)\n        m[idx] = 1.0\n        SITE_MASK[site] = m.to(CFG.DEVICE)\n\nprint(\"site priors:\", {k: int(v.sum().item()) for k,v in SITE_MASK.items()} or \"none\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:01:30.614938Z","iopub.execute_input":"2025-12-27T00:01:30.615135Z","iopub.status.idle":"2025-12-27T00:01:30.928900Z","shell.execute_reply.started":"2025-12-27T00:01:30.615117Z","shell.execute_reply":"2025-12-27T00:01:30.928277Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3) Model: ResNeSt50 (weights must be provided as Kaggle Dataset)","metadata":{}},{"cell_type":"code","source":"# resnest package from dataset (example path); adjust if yours differs\nRESNEST_PKG = Path(\"/kaggle/input/resnest50-fast-package/resnest-0.0.6b20200701\")\nif RESNEST_PKG.exists():\n    import sys\n    sys.path.append(str(RESNEST_PKG))\n    sys.path.append(str(RESNEST_PKG / \"resnest\"))\n\nfrom resnest.torch import resnest50\n\ndef load_resnest50(ckpt_path: Path) -> nn.Module:\n    net = resnest50(pretrained=False)\n    net.fc = nn.Linear(net.fc.in_features, NUM_CLASSES)\n\n    state = torch.load(ckpt_path, map_location=\"cpu\")\n    # remove common prefixes\n    clean = {}\n    for k, v in state.items():\n        if k.startswith(\"model.\"):\n            k = k[6:]\n        if k.startswith(\"module.\"):\n            k = k[7:]\n        clean[k] = v\n\n    net.load_state_dict(clean, strict=True)\n    net.to(CFG.DEVICE).eval()\n    return net\n\nCKPTS = [\n    Path(\"/kaggle/input/kkiller-birdclef-models-public/\"\n         \"birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\")\n]\nnets = [load_resnest50(p) for p in CKPTS]\nprint(\"Ensemble size:\", len(nets))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:03:01.878938Z","iopub.execute_input":"2025-12-27T00:03:01.879597Z","iopub.status.idle":"2025-12-27T00:03:02.467034Z","shell.execute_reply.started":"2025-12-27T00:03:01.879568Z","shell.execute_reply":"2025-12-27T00:03:02.466236Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4) Soundscape table + expected seconds per file","metadata":{}},{"cell_type":"code","source":"files = []\nfor p in TEST_ROOT.glob(\"*.ogg\"):\n    stem = p.stem  # e.g. 20152_SSW_20170805\n    parts = stem.split(\"_\")\n    if len(parts) >= 3:\n        audio_id, site, date = parts[0], parts[1], parts[2]\n    else:\n        audio_id, site, date = stem, \"UNK\", \"UNK\"\n    files.append((stem, audio_id, site, date, p))\n\ndata = pd.DataFrame(files, columns=[\"filename\", \"id\", \"site\", \"date\", \"filepath\"])\nprint(\"soundscapes:\", len(data))\n\nsample_df = None\nif SAMPLE_SUB is not None and SAMPLE_SUB.exists():\n    sample_df = pd.read_csv(SAMPLE_SUB, usecols=[\"row_id\"])\n\ndef seconds_for_file(audio_id: str, site: str):\n    # If sample_submission exists -> follow it exactly (best practice).\n    if sample_df is not None:\n        prefix = f\"{audio_id}_{site}_\"\n        rows = sample_df.loc[sample_df[\"row_id\"].str.startswith(prefix), \"row_id\"]\n        secs = rows.str.split(\"_\").str[-1].astype(int).tolist()\n        secs = sorted(secs)\n        return secs\n    # offline fallback (BirdCLEF 2021 soundscapes are typically 10 min => 5..600 step 5)\n    return list(range(CFG.SEG_SECONDS, 601, CFG.SEG_SECONDS))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:03:02.468600Z","iopub.execute_input":"2025-12-27T00:03:02.469110Z","iopub.status.idle":"2025-12-27T00:03:02.518187Z","shell.execute_reply.started":"2025-12-27T00:03:02.469084Z","shell.execute_reply":"2025-12-27T00:03:02.517380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5) Audio -> mel -> 3ch float tensor","metadata":{}},{"cell_type":"code","source":"def read_audio(path: Path) -> np.ndarray:\n    y, sr = sf.read(path, dtype=\"float32\", always_2d=False)\n    if y.ndim == 2:\n        y = y.mean(axis=1)\n    if sr != CFG.SR:\n        y = lb.resample(y, orig_sr=sr, target_sr=CFG.SR, res_type=\"kaiser_fast\")\n    return y.astype(np.float32)\n\ndef mel_image(y_5s: np.ndarray) -> np.ndarray:\n    # y_5s: shape [SEG_SAMPLES] (pad if needed)\n    if len(y_5s) < CFG.SEG_SAMPLES:\n        y_5s = np.pad(y_5s, (0, CFG.SEG_SAMPLES - len(y_5s)), mode=\"constant\")\n\n    S = lb.feature.melspectrogram(\n        y=y_5s, sr=CFG.SR, n_mels=CFG.N_MELS, fmin=CFG.FMIN, fmax=CFG.FMAX,\n        n_fft=CFG.N_FFT, hop_length=CFG.HOP, power=2.0\n    )\n    S = lb.power_to_db(S).astype(np.float32)\n\n    # z-norm -> minmax -> [0,1], then repeat to 3 channels\n    S = (S - S.mean()) / (S.std() + 1e-6)\n    mn, mx = S.min(), S.max()\n    if mx - mn > 1e-6:\n        S = (S - mn) / (mx - mn)\n    else:\n        S = np.zeros_like(S, dtype=np.float32)\n\n    img = np.stack([S, S, S], axis=0)  # [3, H, W]\n    return img.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:03:02.519190Z","iopub.execute_input":"2025-12-27T00:03:02.519533Z","iopub.status.idle":"2025-12-27T00:03:02.526227Z","shell.execute_reply.started":"2025-12-27T00:03:02.519494Z","shell.execute_reply":"2025-12-27T00:03:02.525474Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6) Postprocess (threshold + optional site mask)","metadata":{}},{"cell_type":"code","source":"# -------------------------\n@torch.inference_mode()\ndef predict_one_file(path: Path, audio_id: str, site: str, secs: list[int]) -> list[str]:\n    y = read_audio(path)\n\n    # build batch of segments in the order of `secs`\n    # segment for second=t is audio[(t-5)*sr : t*sr]\n    xs = []\n    for t in secs:\n        a = (t - CFG.SEG_SECONDS) * CFG.SR\n        b = t * CFG.SR\n        seg = y[a:b] if a < len(y) else np.zeros(0, np.float32)\n        xs.append(mel_image(seg))\n\n    # inference in chunks\n    out_strings = []\n    site_mask = SITE_MASK.get(site, None)\n\n    for start in range(0, len(xs), CFG.BATCH):\n        batch = np.stack(xs[start:start+CFG.BATCH], axis=0)  # [B,3,H,W]\n        xb = torch.from_numpy(batch).to(CFG.DEVICE)\n\n        probs = torch.zeros((xb.size(0), NUM_CLASSES), device=CFG.DEVICE, dtype=torch.float32)\n        for net in nets:\n            logits = net(xb)\n            probs += torch.sigmoid(logits)\n        probs /= len(nets)\n\n        # apply site prior (reduce false positives)\n        if site_mask is not None:\n            probs = probs * site_mask.unsqueeze(0)\n\n        probs_cpu = probs.detach().cpu().numpy()\n        for row in probs_cpu:\n            idx = np.where(row > CFG.THRESH)[0]\n            if idx.size == 0:\n                out_strings.append(\"nocall\")\n            else:\n                # sort by confidence desc\n                idx = idx[np.argsort(-row[idx])]\n                out_strings.append(\" \".join(ID2LABEL[i] for i in idx))\n\n    return out_strings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:03:02.527780Z","iopub.execute_input":"2025-12-27T00:03:02.528016Z","iopub.status.idle":"2025-12-27T00:03:02.541141Z","shell.execute_reply.started":"2025-12-27T00:03:02.527995Z","shell.execute_reply":"2025-12-27T00:03:02.540369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7) Build submission","metadata":{}},{"cell_type":"code","source":"rows = {\"row_id\": [], \"birds\": []}\n\nfor r in data.itertuples(index=False):\n    secs = seconds_for_file(r.id, r.site)\n    preds = predict_one_file(r.filepath, r.id, r.site, secs)\n    assert len(preds) == len(secs)\n\n    for t, birds in zip(secs, preds):\n        rows[\"row_id\"].append(f\"{r.id}_{r.site}_{t}\")\n        rows[\"birds\"].append(birds)\n\nsub = pd.DataFrame(rows)\n\n# If sample_submission exists, match its order strictly\nif sample_df is not None:\n    sub = sample_df.merge(sub, on=\"row_id\", how=\"left\")\n    sub[\"birds\"] = sub[\"birds\"].fillna(\"nocall\")\n\nsub.to_csv(\"submission.csv\", index=False)\nprint(\"Saved: submission.csv | shape:\", sub.shape)\ndisplay(sub.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:03:02.542787Z","iopub.execute_input":"2025-12-27T00:03:02.543079Z","iopub.status.idle":"2025-12-27T00:04:09.822878Z","shell.execute_reply.started":"2025-12-27T00:03:02.543052Z","shell.execute_reply":"2025-12-27T00:04:09.821941Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8) Offline evaluation (approx row-wise micro F1)","metadata":{}},{"cell_type":"code","source":"def rowwise_micro_f1(true_birds, pred_birds) -> float:\n    tp = fp = fn = 0\n    for yt, yp in zip(true_birds, pred_birds):\n        yt = \"nocall\" if not isinstance(yt, str) else yt\n        yp = \"nocall\" if not isinstance(yp, str) else yp\n        tset = set() if yt == \"nocall\" else set(yt.split())\n        pset = set() if yp == \"nocall\" else set(yp.split())\n        tp += len(tset & pset)\n        fp += len(pset - tset)\n        fn += len(tset - pset)\n    if tp + fp + fn == 0:\n        return 0.0\n    prec = tp / (tp + fp + 1e-8)\n    rec  = tp / (tp + fn + 1e-8)\n    return 0.0 if (prec + rec) == 0 else 2 * prec * rec / (prec + rec)\n\nif TARGET.exists() and sample_df is None:\n    gt = pd.read_csv(TARGET, usecols=[\"row_id\", \"birds\"])\n    merged = gt.merge(sub, on=\"row_id\", how=\"left\", suffixes=(\"_true\", \"_pred\"))\n    score = rowwise_micro_f1(merged[\"birds_true\"].tolist(),\n                             merged[\"birds_pred\"].fillna(\"nocall\").tolist())\n    print(f\"Offline row-wise micro F1: {score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T00:04:09.823721Z","iopub.execute_input":"2025-12-27T00:04:09.824003Z","iopub.status.idle":"2025-12-27T00:04:09.871855Z","shell.execute_reply.started":"2025-12-27T00:04:09.823973Z","shell.execute_reply":"2025-12-27T00:04:09.871317Z"}},"outputs":[],"execution_count":null}]}