{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🐦 BirdCLEF 2021 — детекция птиц по soundscapes с ResNeSt50 (inference-only, fast & pretty)","metadata":{}},{"cell_type":"code","source":"import os, math, gc, sys\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\n\nimport soundfile as sf\nimport librosa as lb\n\nimport torch\nimport torch.nn as nn\n\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\n\nprint(\"torch:\", torch.__version__, \"| cuda:\", torch.cuda.is_available())\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(\"DEVICE:\", DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:33:18.103505Z","iopub.execute_input":"2025-12-27T14:33:18.103731Z","iopub.status.idle":"2025-12-27T14:33:23.594472Z","shell.execute_reply.started":"2025-12-27T14:33:18.103711Z","shell.execute_reply":"2025-12-27T14:33:23.593653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SR = 32000\nWIN_SEC = 5\nWIN_SAMPLES = SR * WIN_SEC\n\nTOTAL_SEC = 600\nN_WINDOWS = TOTAL_SEC // WIN_SEC          # 120\nTOTAL_SAMPLES = TOTAL_SEC * SR\n\n# --- батчинг окон на GPU ---\nBATCH_WINDOWS = 16                         \nUSE_AMP = False                            \n\n# --- пороги  ---\nTHRESH_GRID = [0.10, 0.15, 0.20, 0.25, 0.30, 0.35]\nDEFAULT_THR = 0.10                         \n\n\nN_MELS = 128\nFMIN = 0\nFMAX = SR // 2\nN_FFT = SR // 10                           # 3200\nHOP = SR // 40                             # 800\nRES_TYPE = \"kaiser_fast\"\n\nprint(\"Mel params:\", dict(N_MELS=N_MELS, N_FFT=N_FFT, HOP=HOP, FMIN=FMIN, FMAX=FMAX))\n\nDATA_ROOT = Path(\"../input/birdclef-2021\")\n\nTEST_AUDIO_ROOT = DATA_ROOT / \"test_soundscapes\"\nSAMPLE_SUB_PATH = DATA_ROOT / \"sample_submission.csv\"\nTARGET_PATH = None\n\nif not len(list(TEST_AUDIO_ROOT.glob(\"*.ogg\"))):\n    TEST_AUDIO_ROOT = DATA_ROOT / \"train_soundscapes\"\n    SAMPLE_SUB_PATH = None\n    TARGET_PATH = DATA_ROOT / \"train_soundscape_labels.csv\"\n\nOFFLINE = (TARGET_PATH is not None) and Path(TARGET_PATH).exists()\n\nprint(\"Audio root:\", TEST_AUDIO_ROOT)\nprint(\"SAMPLE_SUB_PATH:\", SAMPLE_SUB_PATH)\nprint(\"TARGET_PATH:\", TARGET_PATH)\nprint(\"OFFLINE:\", OFFLINE)\n\ndata = pd.DataFrame(\n    [(p.stem, *p.stem.split(\"_\"), p) for p in TEST_AUDIO_ROOT.glob(\"*.ogg\")],\n    columns=[\"filename\", \"id\", \"site\", \"date\", \"filepath\"],\n)\nprint(\"soundscapes files:\", len(data))\ndisplay(data.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:33:23.596191Z","iopub.execute_input":"2025-12-27T14:33:23.596549Z","iopub.status.idle":"2025-12-27T14:33:23.638940Z","shell.execute_reply.started":"2025-12-27T14:33:23.596525Z","shell.execute_reply":"2025-12-27T14:33:23.638252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_audio_info(path: Path):\n    y, sr0 = sf.read(path, dtype=\"float32\")\n    if y.ndim > 1:\n        y = y.mean(axis=1)\n    dur = len(y) / sr0\n    return dur, sr0, float(np.abs(y).mean()), float(np.abs(y).max())\n\nsample_paths = data.sample(min(5, len(data)), random_state=42)[\"filepath\"].tolist()\ninfos = []\nfor p in sample_paths:\n    dur, sr0, mean_abs, max_abs = read_audio_info(p)\n    infos.append((p.name, dur, sr0, mean_abs, max_abs))\n\neda_df = pd.DataFrame(infos, columns=[\"file\",\"duration_s\",\"sr\",\"mean_abs\",\"max_abs\"])\ndisplay(eda_df)\n\n# %%\n# график распределения длительностей\nplt.figure(figsize=(7,4))\nplt.hist(eda_df[\"duration_s\"].values, bins=10)\nplt.title(\"Durations (sample)\")\nplt.xlabel(\"seconds\")\nplt.ylabel(\"count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:35.936144Z","iopub.execute_input":"2025-12-27T14:34:35.937008Z","iopub.status.idle":"2025-12-27T14:34:39.330975Z","shell.execute_reply.started":"2025-12-27T14:34:35.936964Z","shell.execute_reply":"2025-12-27T14:34:39.330358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"p0 = data.iloc[0].filepath\ny, sr0 = sf.read(p0, dtype=\"float32\")\nif y.ndim > 1: y = y.mean(axis=1)\n\n# ресемплим до SR для честной визуализации\nif sr0 != SR:\n    y_rs = lb.resample(y, orig_sr=sr0, target_sr=SR, res_type=RES_TYPE)\nelse:\n    y_rs = y\n\nsec_show = 10\nplt.figure(figsize=(10,3))\nplt.plot(y_rs[:sec_show*SR])\nplt.title(f\"Waveform (first {sec_show}s): {p0.name}\")\nplt.xlabel(\"samples\")\nplt.ylabel(\"amplitude\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:39.332052Z","iopub.execute_input":"2025-12-27T14:34:39.332276Z","iopub.status.idle":"2025-12-27T14:34:40.128906Z","shell.execute_reply.started":"2025-12-27T14:34:39.332255Z","shell.execute_reply":"2025-12-27T14:34:40.128250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n\n    _min, _max = X.min(), X.max()\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n    return V\n\ndef audio_to_image_librosa(seg: np.ndarray) -> np.ndarray:\n    melspec = lb.feature.melspectrogram(\n        y=seg,\n        sr=SR,\n        n_mels=N_MELS,\n        fmin=FMIN,\n        fmax=FMAX,\n        n_fft=N_FFT,\n        hop_length=HOP,\n        win_length=N_FFT,\n        power=2.0,\n        center=True,\n        pad_mode=\"reflect\",\n    )\n    melspec = lb.power_to_db(melspec).astype(np.float32)\n    img = mono_to_color(melspec).astype(np.float32) / 255.0\n    img = np.stack([img, img, img], axis=0)  # [3, 128, T]\n    return img\n\n# берем один 5сек сегмент\nseg0 = y_rs[:WIN_SAMPLES].astype(np.float32, copy=False)\nimg0 = audio_to_image_librosa(seg0)\nprint(\"Image shape:\", img0.shape, \"(expect [3,128,~201])\")\n\nplt.figure(figsize=(10,4))\nplt.imshow(np.transpose(img0, (1,2,0)))\nplt.title(\"Mel-spectrogram as image (5 sec window)\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:40.129824Z","iopub.execute_input":"2025-12-27T14:34:40.130122Z","iopub.status.idle":"2025-12-27T14:34:52.993827Z","shell.execute_reply.started":"2025-12-27T14:34:40.130090Z","shell.execute_reply":"2025-12-27T14:34:52.993182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(DATA_ROOT / \"train_metadata.csv\")\nlabels_sorted = sorted(df_train[\"primary_label\"].unique())\nLABEL2IDX = {lb: i for i, lb in enumerate(labels_sorted)}\nIDX2LABEL = labels_sorted\nNUM_CLASSES = len(IDX2LABEL)\nprint(\"NUM_CLASSES:\", NUM_CLASSES)\n\n# топ-20 самых частых видов (по train_metadata)\ntop_counts = df_train[\"primary_label\"].value_counts().head(20)\nplt.figure(figsize=(10,4))\nplt.bar(range(len(top_counts)), top_counts.values)\nplt.title(\"Top-20 most frequent labels in train_metadata\")\nplt.xticks(range(len(top_counts)), top_counts.index, rotation=90)\nplt.ylabel(\"count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:52.995149Z","iopub.execute_input":"2025-12-27T14:34:52.995468Z","iopub.status.idle":"2025-12-27T14:34:53.530547Z","shell.execute_reply.started":"2025-12-27T14:34:52.995447Z","shell.execute_reply":"2025-12-27T14:34:53.530006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sys.path.append(\"/kaggle/input/resnest50-fast-package/resnest-0.0.6b20200701\")\nsys.path.append(\"/kaggle/input/resnest50-fast-package/resnest-0.0.6b20200701/resnest\")\nfrom resnest.torch import resnest50\n\ndef load_net_resnest(checkpoint_path: Path) -> nn.Module:\n    net = resnest50(pretrained=False)\n    net.fc = nn.Linear(net.fc.in_features, NUM_CLASSES)\n\n    state = torch.load(checkpoint_path, map_location=\"cpu\", weights_only=False)\n\n    # иногда веса с префиксом model.\n    if isinstance(state, dict):\n        for k in list(state.keys()):\n            if k.startswith(\"model.\"):\n                state[k[6:]] = state.pop(k)\n\n    net.load_state_dict(state, strict=True)\n    net.to(DEVICE).eval()\n    return net\n\ncheckpoint_paths = [\n    Path(\"/kaggle/input/kkiller-birdclef-models-public/birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\")\n]\nnets = [load_net_resnest(p) for p in checkpoint_paths]\nprint(\"Ensemble size:\", len(nets))\nprint(\"Head:\", nets[0].fc)\n\n# sanity forward\nwith torch.no_grad():\n    xb = torch.from_numpy(np.stack([img0])).to(DEVICE)  # [1,3,128,T]\n    out = nets[0](xb)\nprint(\"logits shape:\", tuple(out.shape))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:53.531315Z","iopub.execute_input":"2025-12-27T14:34:53.531514Z","iopub.status.idle":"2025-12-27T14:34:56.060421Z","shell.execute_reply.started":"2025-12-27T14:34:53.531495Z","shell.execute_reply":"2025-12-27T14:34:56.059686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_audio_exact(filepath: Path) -> np.ndarray:\n    audio, orig_sr = sf.read(filepath, dtype=\"float32\")\n    if audio.ndim > 1:\n        audio = audio.mean(axis=1)\n    if orig_sr != SR:\n        audio = lb.resample(audio, orig_sr=orig_sr, target_sr=SR, res_type=RES_TYPE)\n\n    # ровно 600 сек\n    if len(audio) < TOTAL_SAMPLES:\n        audio = np.pad(audio, (0, TOTAL_SAMPLES - len(audio)))\n    else:\n        audio = audio[:TOTAL_SAMPLES]\n    return audio.astype(np.float32, copy=False)\n\ndef decode_row(p_row: torch.Tensor, thresh: float) -> str:\n    # как в твоей логике: берём top-n, где n = count(prob > thresh)\n    sorted_idx = torch.argsort(p_row, descending=True)\n    n = int((p_row > thresh).sum().item())\n    if n <= 0:\n        return \"nocall\"\n    return \" \".join(IDX2LABEL[i] for i in sorted_idx[:n].tolist())\n\n@torch.inference_mode()\ndef predict_soundscape(filepath: Path, thresh=DEFAULT_THR, batch_windows=BATCH_WINDOWS):\n    audio = load_audio_exact(filepath)  # [600*SR]\n    out = []\n    batch_imgs = []\n\n    for w in range(N_WINDOWS):\n        start = w * WIN_SAMPLES\n        seg = audio[start:start+WIN_SAMPLES]\n        img = audio_to_image_librosa(seg)\n        batch_imgs.append(img)\n\n        if len(batch_imgs) == batch_windows:\n            xb = torch.from_numpy(np.stack(batch_imgs)).to(DEVICE)\n            if USE_AMP and DEVICE == \"cuda\":\n                with torch.autocast(device_type=\"cuda\", enabled=True):\n                    p = 0\n                    for net in nets:\n                        p = p + torch.sigmoid(net(xb).float())\n                    p = p / len(nets)\n            else:\n                p = 0\n                for net in nets:\n                    p = p + torch.sigmoid(net(xb))\n                p = p / len(nets)\n\n            p = p.detach().cpu()\n            out.extend([decode_row(p[i], thresh) for i in range(p.size(0))])\n\n            batch_imgs.clear()\n            del xb, p\n            if DEVICE == \"cuda\":\n                torch.cuda.empty_cache()\n\n    # хвост\n    if batch_imgs:\n        xb = torch.from_numpy(np.stack(batch_imgs)).to(DEVICE)\n        p = 0\n        for net in nets:\n            p = p + torch.sigmoid(net(xb))\n        p = (p / len(nets)).detach().cpu()\n        out.extend([decode_row(p[i], thresh) for i in range(p.size(0))])\n\n        batch_imgs.clear()\n        del xb, p\n        if DEVICE == \"cuda\":\n            torch.cuda.empty_cache()\n\n    return out  # list length 120\n\n# быстрый прогон на 1 файле\npred0 = predict_soundscape(data.iloc[0].filepath, thresh=DEFAULT_THR, batch_windows=16)\nprint(\"windows:\", len(pred0), \"| first 10:\", pred0[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:56.061200Z","iopub.execute_input":"2025-12-27T14:34:56.061406Z","iopub.status.idle":"2025-12-27T14:34:58.832504Z","shell.execute_reply.started":"2025-12-27T14:34:56.061386Z","shell.execute_reply":"2025-12-27T14:34:58.831381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_submission_df(meta: pd.DataFrame, preds: list[list[str]]) -> pd.DataFrame:\n    rows = {\"row_id\": [], \"birds\": []}\n    for r, pp in zip(meta.itertuples(index=False), preds):\n        for i, birds in enumerate(pp, start=1):\n            sec = i * WIN_SEC\n            rows[\"row_id\"].append(f\"{r.id}_{r.site}_{sec}\")\n            rows[\"birds\"].append(birds)\n    out = pd.DataFrame(rows)\n\n    # если есть sample_submission, подгоняем row_id\n    if SAMPLE_SUB_PATH is not None and Path(SAMPLE_SUB_PATH).exists():\n        sample = pd.read_csv(SAMPLE_SUB_PATH, usecols=[\"row_id\"])\n        out = sample.merge(out, on=\"row_id\", how=\"left\")\n        out[\"birds\"] = out[\"birds\"].fillna(\"nocall\")\n    return out\n\ndef run_full_inference(meta_df: pd.DataFrame, thresh=DEFAULT_THR, batch_windows=16) -> pd.DataFrame:\n    preds = []\n    for r in tqdm(list(meta_df.itertuples(index=False)), desc=\"Soundscapes\"):\n        preds.append(predict_soundscape(r.filepath, thresh=thresh, batch_windows=batch_windows))\n    return build_submission_df(meta_df, preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:58.833204Z","iopub.execute_input":"2025-12-27T14:34:58.833444Z","iopub.status.idle":"2025-12-27T14:34:58.843991Z","shell.execute_reply.started":"2025-12-27T14:34:58.833420Z","shell.execute_reply":"2025-12-27T14:34:58.843432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rowwise_micro_f1(true_birds, pred_birds) -> float:\n    tp = fp = fn = 0\n    for yt, yp in zip(true_birds, pred_birds):\n        if isinstance(yt, float) and math.isnan(yt): yt = \"nocall\"\n        if isinstance(yp, float) and math.isnan(yp): yp = \"nocall\"\n        tset = set([] if yt == \"nocall\" else yt.split())\n        pset = set([] if yp == \"nocall\" else yp.split())\n        tp += len(tset & pset)\n        fp += len(pset - tset)\n        fn += len(tset - pset)\n    if tp + fp + fn == 0:\n        return 0.0\n    prec = tp / (tp + fp + 1e-8)\n    rec  = tp / (tp + fn + 1e-8)\n    return 0.0 if (prec + rec) == 0 else (2 * prec * rec / (prec + rec))\n\nbest_thr = DEFAULT_THR\nbest_score = None\nthr_scores = []\n\nif OFFLINE:\n    labels_df = pd.read_csv(TARGET_PATH)[[\"row_id\",\"birds\"]]\n    for thr in THRESH_GRID:\n        gc.collect()\n        if DEVICE == \"cuda\":\n            torch.cuda.empty_cache()\n\n        pred_df = run_full_inference(data, thresh=thr, batch_windows=16)\n        m = labels_df.merge(pred_df, on=\"row_id\", how=\"left\", suffixes=(\"_true\",\"_pred\"))\n\n        score = rowwise_micro_f1(\n            m[\"birds_true\"].tolist(),\n            m[\"birds_pred\"].fillna(\"nocall\").tolist()\n        )\n        thr_scores.append((thr, score))\n        print(f\"thr={thr:.2f} | offline rowwise micro F1 = {score:.4f}\")\n\n        if (best_score is None) or (score > best_score):\n            best_score = score\n            best_thr = thr\n\n    print(\"\\n✅ Best thr:\", best_thr, \"| score:\", best_score)\n\n    # график F1 от порога\n    xs = [t for t,_ in thr_scores]\n    ys = [s for _,s in thr_scores]\n    plt.figure(figsize=(7,4))\n    plt.plot(xs, ys, marker=\"o\")\n    plt.title(\"Offline row-wise micro F1 vs threshold\")\n    plt.xlabel(\"threshold\")\n    plt.ylabel(\"F1\")\n    plt.grid(True)\n    plt.show()\nelse:\n    print(\"OFFLINE=False -> пропускаем подбор порога (чтобы hidden/test не падал).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:34:58.846233Z","iopub.execute_input":"2025-12-27T14:34:58.846461Z","iopub.status.idle":"2025-12-27T14:40:09.294815Z","shell.execute_reply.started":"2025-12-27T14:34:58.846438Z","shell.execute_reply":"2025-12-27T14:40:09.294204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"thr_for_analysis = best_thr if (best_score is not None) else DEFAULT_THR\nsub_df = run_full_inference(data, thresh=thr_for_analysis, batch_windows=16)\ndisplay(sub_df.head())\n\n# %%\n# доля nocall\nis_nocall = (sub_df[\"birds\"] == \"nocall\").values\nnocall_ratio = is_nocall.mean()\nprint(\"nocall ratio:\", round(float(nocall_ratio), 4))\n\nplt.figure(figsize=(6,4))\nplt.hist(is_nocall.astype(int), bins=3)\nplt.title(\"Nocall distribution (0=bird(s), 1=nocall)\")\nplt.xlabel(\"value\")\nplt.ylabel(\"count\")\nplt.show()\n\n# %%\n# сколько птиц в окне (если nocall -> 0)\ncounts = []\nfor s in sub_df[\"birds\"].values:\n    if s == \"nocall\" or not isinstance(s, str):\n        counts.append(0)\n    else:\n        counts.append(len(s.split()))\n\nplt.figure(figsize=(7,4))\nplt.hist(counts, bins=20)\nplt.title(\"How many birds predicted per 5-sec window\")\nplt.xlabel(\"num birds\")\nplt.ylabel(\"count\")\nplt.show()\n\n# %%\n# топ-виды по частоте (исключая nocall)\nall_pred_species = []\nfor s in sub_df[\"birds\"].values:\n    if isinstance(s, str) and s != \"nocall\":\n        all_pred_species.extend(s.split())\n\ncnt = Counter(all_pred_species)\ntop20 = cnt.most_common(20)\nprint(\"Top-20 predicted species:\")\ndisplay(pd.DataFrame(top20, columns=[\"species\",\"count\"]))\n\nplt.figure(figsize=(10,4))\nplt.bar(range(len(top20)), [c for _,c in top20])\nplt.title(\"Top-20 predicted species (frequency)\")\nplt.xticks(range(len(top20)), [sp for sp,_ in top20], rotation=90)\nplt.ylabel(\"count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:40:09.295696Z","iopub.execute_input":"2025-12-27T14:40:09.295978Z","iopub.status.idle":"2025-12-27T14:41:00.685175Z","shell.execute_reply.started":"2025-12-27T14:40:09.295943Z","shell.execute_reply":"2025-12-27T14:41:00.684540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = sub_df.copy()\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"✅ Saved submission.csv:\", submission.shape)\ndisplay(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T14:41:00.687176Z","iopub.execute_input":"2025-12-27T14:41:00.687388Z","iopub.status.idle":"2025-12-27T14:41:00.704423Z","shell.execute_reply.started":"2025-12-27T14:41:00.687368Z","shell.execute_reply":"2025-12-27T14:41:00.703923Z"}},"outputs":[],"execution_count":null}]}