{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":129329,"databundleVersionId":15996945},{"sourceType":"datasetVersion","sourceId":15924303,"datasetId":10212633,"databundleVersionId":16881164},{"sourceType":"datasetVersion","sourceId":15419678,"datasetId":9864137,"databundleVersionId":16337168},{"sourceType":"datasetVersion","sourceId":15411593,"datasetId":9859174,"databundleVersionId":16327938},{"sourceType":"datasetVersion","sourceId":16382163,"datasetId":10501897,"databundleVersionId":17375820},{"sourceType":"datasetVersion","sourceId":15375708,"datasetId":9835500,"databundleVersionId":16287978},{"sourceType":"datasetVersion","sourceId":15696471,"datasetId":10054432,"databundleVersionId":16635312},{"sourceType":"datasetVersion","sourceId":15653238,"datasetId":10021395,"databundleVersionId":16589328},{"sourceType":"datasetVersion","sourceId":15166537,"datasetId":9712127,"databundleVersionId":16058239},{"sourceType":"datasetVersion","sourceId":16097880,"datasetId":9872649,"databundleVersionId":17068561},{"sourceType":"datasetVersion","sourceId":15437640,"datasetId":9875157,"databundleVersionId":16357325},{"sourceType":"datasetVersion","sourceId":15524136,"datasetId":9932123,"databundleVersionId":16451284},{"sourceType":"datasetVersion","sourceId":15842124,"datasetId":10155577,"databundleVersionId":16792642},{"sourceType":"modelInstanceVersion","sourceId":32637,"databundleVersionId":8261530,"modelInstanceId":26649,"modelId":37756},{"sourceType":"modelInstanceVersion","sourceId":516989,"databundleVersionId":13353982,"modelInstanceId":404337,"modelId":319},{"sourceType":"modelInstanceVersion","sourceId":542750,"databundleVersionId":13512426,"modelInstanceId":418187,"modelId":319},{"sourceType":"kernelVersion","sourceId":299467559},{"sourceType":"kernelVersion","sourceId":306125702},{"sourceType":"kernelVersion","sourceId":320601599},{"sourceType":"kernelVersion","sourceId":320892641}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import subprocess, sys, os\nfrom pathlib import Path\nINPUT_ROOT = Path(\"/kaggle/input\")\ndef find_wheel(pattern):\n    for p in INPUT_ROOT.rglob(pattern):\n        return p\n    raise FileNotFoundError(pattern)\nONNX_WHL = Path(\"/kaggle/input/datasets/rishikeshjani/perch-onnx-for-birdclef-2026/onnxruntime-1.24.4-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl\")\nif ONNX_WHL.exists():\n    subprocess.run([sys.executable, \"-m\", \"pip\", \"install\", \"-q\", \"--no-deps\", str(ONNX_WHL)], check=True)\n    print(\"ONNX Runtime installed\")\nsubprocess.run([sys.executable, \"-m\", \"pip\", \"install\", \"-q\", \"--no-deps\",\n                str(find_wheel(\"tensorboard-2.20.0-*.whl\"))], check=True)\nsubprocess.run([sys.executable, \"-m\", \"pip\", \"install\", \"-q\", \"--no-deps\",\n                str(find_wheel(\"tensorflow-2.20.0-*.whl\"))], check=True)\nprint(\"TF 2.20 installed\")\ntry:\n    import onnxruntime as ort\n    _ONNX_AVAILABLE = True\n    print(\"ONNX Runtime available\")\nexcept ImportError:\n    _ONNX_AVAILABLE = False\n    print(\"ONNX not available, falling back to TF\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport os\nimport numpy as np\nimport torch\ndef seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\nseed_everything(42)\nprint(\"Global random seed set to 42\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MODE = \"submit\"\nassert MODE in {\"train\", \"submit\"}\nprint(\"MODE =\", MODE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, re, gc, time, warnings\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"3\"\nwarnings.filterwarnings(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\nimport soundfile as sf\nimport tensorflow as tf\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.auto import tqdm\n\ntf.experimental.numpy.experimental_enable_numpy_behavior()\ntry: tf.config.set_visible_devices([], \"GPU\")\nexcept: pass\n\n_WALL_START = time.time()\n\nBASE      = Path(\"/kaggle/input/competitions/birdclef-2026\")\nMODEL_DIR = Path(\"/kaggle/input/models/google/bird-vocalization-classifier/tensorflow2/perch_v2_cpu/1\")\nWORK_DIR  = Path(\"/kaggle/working/cache\")\nWORK_DIR.mkdir(parents=True, exist_ok=True)\n\nSR             = 32_000\nWINDOW_SEC     = 5\nWINDOW_SAMPLES = SR * WINDOW_SEC\nFILE_SAMPLES   = 60 * SR\nN_WINDOWS      = 12\n\nCFG = {\n    \"batch_files\": 16,\n    \"oof_n_splits\": 5   if MODE == \"train\" else 3,\n    \"dryrun_n_files\": 20 if MODE == \"train\" else 0,\n    \"run_oof\": MODE == \"train\",\n    \"verbose\": MODE == \"train\",\n    \"proto_ssm_train\": {\n        \"n_epochs\":        80  if MODE == \"train\" else 40,\n        \"lr\":              8e-4,\n        \"weight_decay\":    1e-3,\n        \"val_ratio\":       0.15,\n        \"patience\":        20  if MODE == \"train\" else 8,\n        \"pos_weight_cap\":  25.0,\n        \"distill_weight\":  0.15,\n        \"proto_margin\":    0.15,\n        \"label_smoothing\": 0.03,\n        \"oof_n_splits\":    5   if MODE == \"train\" else 3,\n        \"mixup_alpha\":     0.4,\n        \"focal_gamma\":     2.5,\n        \"swa_start_frac\":  0.65,\n        \"swa_lr\":          4e-4,\n        \"use_cosine_restart\": True,\n        \"restart_period\":  20,\n        \"d_model\":         128,   # keep 128 — right size for 59 files\n    },\n    \"residual_ssm\": {\n        \"d_model\": 128, \"d_state\": 16, \"n_ssm_layers\": 2,\n        \"dropout\": 0.1, \"correction_weight\": 0.35,\n        \"n_epochs\": 40  if MODE == \"train\" else 20,\n        \"lr\": 8e-4,\n        \"patience\": 12  if MODE == \"train\" else 6,\n    },\n    \"mlp_params\": {\n        \"hidden_layer_sizes\": (128, 64),   # proven size for 59 files\n        \"activation\": \"relu\",\n        \"max_iter\": 500  if MODE == \"train\" else 200,\n        \"early_stopping\": True,\n        \"validation_fraction\": 0.15,\n        \"n_iter_no_change\": 20  if MODE == \"train\" else 10,\n        \"random_state\": 42,\n        \"learning_rate_init\": 5e-4,\n        \"alpha\": 0.005,\n    },\n    \"mlp_pca_dim\": 64,    # proven pca for 59 files\n    \"tta_shifts\": [0, 1, -1, 2, -2],  # TTA on BOTH train and test\n}\nprint(\"V86 CFG loaded\")\nprint(f\"  d_model={CFG['proto_ssm_train']['d_model']}  \"\n      f\"mlp={CFG['mlp_params']['hidden_layer_sizes']}  \"\n      f\"pca={CFG['mlp_pca_dim']}  \"\n      f\"tta={CFG['tta_shifts']}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy          = pd.read_csv(BASE / \"taxonomy.csv\")\nsample_sub        = pd.read_csv(BASE / \"sample_submission.csv\")\nsoundscape_labels = pd.read_csv(BASE / \"train_soundscapes_labels.csv\")\n\nPRIMARY_LABELS = sample_sub.columns[1:].tolist()\nN_CLASSES      = len(PRIMARY_LABELS)\nlabel_to_idx   = {c: i for i, c in enumerate(PRIMARY_LABELS)}\n\nFNAME_RE = re.compile(r\"BC2026_(?:Train|Test)_(\\d+)_(S\\d+)_(\\d{8})_(\\d{6})\\.ogg\")\n\ndef parse_fname(name):\n    m = FNAME_RE.match(name)\n    if not m: return {\"site\": \"unknown\", \"hour_utc\": -1}\n    _, site, _, hms = m.groups()\n    return {\"site\": site, \"hour_utc\": int(hms[:2])}\n\ndef union_labels(series):\n    out = set()\n    for x in series:\n        if pd.notna(x):\n            for t in str(x).split(\";\"):\n                t = t.strip()\n                if t: out.add(t)\n    return sorted(out)\n\nsc = (soundscape_labels\n      .groupby([\"filename\", \"start\", \"end\"])[\"primary_label\"]\n      .apply(union_labels)\n      .reset_index(name=\"label_list\"))\n\nsc[\"end_sec\"] = pd.to_timedelta(sc[\"end\"]).dt.total_seconds().astype(int)\nsc[\"row_id\"]  = sc[\"filename\"].str.replace(\".ogg\", \"\", regex=False) + \"_\" + sc[\"end_sec\"].astype(str)\n\n_meta = sc[\"filename\"].apply(parse_fname).apply(pd.Series)\nsc = pd.concat([sc, _meta], axis=1)\n\nY_SC = np.zeros((len(sc), N_CLASSES), dtype=np.uint8)\nfor i, lbls in enumerate(sc[\"label_list\"]):\n    for lbl in lbls:\n        if lbl in label_to_idx:\n            Y_SC[i, label_to_idx[lbl]] = 1\n\nwindows_per_file = sc.groupby(\"filename\").size()\nfull_files = sorted(windows_per_file[windows_per_file == N_WINDOWS].index.tolist())\nsc[\"fully_labeled\"] = sc[\"filename\"].isin(full_files)\n\nfull_rows = (sc[sc[\"fully_labeled\"]]\n             .sort_values([\"filename\", \"end_sec\"])\n             .reset_index(drop=False))\nY_FULL = Y_SC[full_rows[\"index\"].to_numpy()]\n\nprint(f\"Classes: {N_CLASSES} | Fully-labeled files: {len(full_files)}\")\nprint(f\"Full-file windows: {len(full_rows)} | Active classes: {int((Y_FULL.sum(0) > 0).sum())}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"birdclassifier = tf.saved_model.load(str(MODEL_DIR))\ninfer_fn       = birdclassifier.signatures[\"serving_default\"]\nONNX_PERCH_PATH = Path(\"/kaggle/input/datasets/rishikeshjani/perch-onnx-for-birdclef-2026/perch_v2.onnx\")\nUSE_ONNX = _ONNX_AVAILABLE and ONNX_PERCH_PATH.exists()\nif USE_ONNX:\n    _so = ort.SessionOptions()\n    _so.intra_op_num_threads = 4\n    ONNX_SESSION    = ort.InferenceSession(str(ONNX_PERCH_PATH), sess_options=_so,\n                                            providers=[\"CPUExecutionProvider\"])\n    ONNX_INPUT_NAME = ONNX_SESSION.get_inputs()[0].name\n    ONNX_OUT_MAP    = {o.name: i for i, o in enumerate(ONNX_SESSION.get_outputs())}\n    print(\"Using ONNX Perch (150x faster)\")\nelse:\n    print(\"Using TF SavedModel Perch\")\nbc_labels = (pd.read_csv(MODEL_DIR / \"assets\" / \"labels.csv\")\n             .reset_index()\n             .rename(columns={\"index\": \"bc_index\", \"inat2024_fsd50k\": \"scientific_name\"}))\nNO_LABEL = len(bc_labels)\nmapping = (taxonomy\n           .merge(bc_labels.rename(columns={\"scientific_name\": \"scientific_name\"}),\n                  on=\"scientific_name\", how=\"left\"))\nmapping[\"bc_index\"] = mapping[\"bc_index\"].fillna(NO_LABEL).astype(int)\nlbl2bc = mapping.set_index(\"primary_label\")[\"bc_index\"]\nBC_INDICES    = np.array([int(lbl2bc.loc[c]) for c in PRIMARY_LABELS], dtype=np.int32)\nMAPPED_MASK   = BC_INDICES != NO_LABEL\nMAPPED_POS    = np.where(MAPPED_MASK)[0].astype(np.int32)\nMAPPED_BC_IDX = BC_INDICES[MAPPED_MASK].astype(np.int32)\nprint(f\"Mapped: {MAPPED_MASK.sum()} / {N_CLASSES} species have a Perch logit\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re as _re\nUNMAPPED_POS  = np.where(~MAPPED_MASK)[0].astype(np.int32)\nCLASS_NAME_MAP = taxonomy.set_index(\"primary_label\")[\"class_name\"].to_dict()\nTEXTURE_TAXA   = {\"Amphibia\", \"Insecta\"}\nproxy_map = {}\nunmapped_df = (taxonomy[taxonomy[\"primary_label\"]\n               .isin([PRIMARY_LABELS[i] for i in UNMAPPED_POS])]\n               .copy())\nfor _, row in unmapped_df.iterrows():\n    target = row[\"primary_label\"]\n    sci    = str(row[\"scientific_name\"])\n    genus  = sci.split()[0]\n    hits = bc_labels[\n        bc_labels[\"scientific_name\"]\n        .astype(str)\n        .str.match(rf\"^{_re.escape(genus)}\\s\", na=False)\n    ]\n    if len(hits) > 0:\n        proxy_map[label_to_idx[target]] = hits[\"bc_index\"].astype(int).tolist()\nPROXY_TAXA = {\"Amphibia\", \"Insecta\", \"Aves\"}\nproxy_map  = {\n    idx: bc_idxs\n    for idx, bc_idxs in proxy_map.items()\n    if CLASS_NAME_MAP.get(PRIMARY_LABELS[idx]) in PROXY_TAXA\n}\nprint(f\"Unmapped: {len(UNMAPPED_POS)} | With proxy: {len(proxy_map)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import concurrent.futures\ndef read_60s(path):\n    y, sr = sf.read(path, dtype=\"float32\", always_2d=False)\n    if y.ndim == 2: y = y.mean(axis=1)\n    if len(y) < FILE_SAMPLES: y = np.pad(y, (0, FILE_SAMPLES - len(y)))\n    else:                      y = y[:FILE_SAMPLES]\n    return y\ndef run_perch(paths, batch_files=16, verbose=True):\n    paths  = [Path(p) for p in paths]\n    n_rows = len(paths) * N_WINDOWS\n    row_ids   = np.empty(n_rows, dtype=object)\n    filenames = np.empty(n_rows, dtype=object)\n    sites     = np.empty(n_rows, dtype=object)\n    hours     = np.zeros(n_rows, dtype=np.int16)\n    scores    = np.zeros((n_rows, N_CLASSES), dtype=np.float32)\n    embs      = np.zeros((n_rows, 1536),      dtype=np.float32)\n    wr  = 0\n    itr = tqdm(range(0, len(paths), batch_files), desc=\"Perch\") if verbose else range(0, len(paths), batch_files)\n    with concurrent.futures.ThreadPoolExecutor(max_workers=4) as io_executor:\n        next_paths   = paths[0:batch_files]\n        future_audio = [io_executor.submit(read_60s, p) for p in next_paths]\n        for start in itr:\n            batch_paths  = next_paths\n            batch_n      = len(batch_paths)\n            batch_audio  = [f.result() for f in future_audio]\n            next_start = start + batch_files\n            if next_start < len(paths):\n                next_paths   = paths[next_start:next_start + batch_files]\n                future_audio = [io_executor.submit(read_60s, p) for p in next_paths]\n            x  = np.empty((batch_n * N_WINDOWS, WINDOW_SAMPLES), dtype=np.float32)\n            br = wr\n            for bi, path in enumerate(batch_paths):\n                y    = batch_audio[bi]\n                meta = parse_fname(path.name)\n                stem = path.stem\n                x[bi * N_WINDOWS:(bi + 1) * N_WINDOWS] = y.reshape(N_WINDOWS, WINDOW_SAMPLES)\n                row_ids  [wr:wr + N_WINDOWS] = [f\"{stem}_{t}\" for t in range(5, 65, 5)]\n                filenames[wr:wr + N_WINDOWS] = path.name\n                sites    [wr:wr + N_WINDOWS] = meta[\"site\"]\n                hours    [wr:wr + N_WINDOWS] = meta[\"hour_utc\"]\n                wr += N_WINDOWS\n            if USE_ONNX:\n                outs   = ONNX_SESSION.run(None, {ONNX_INPUT_NAME: x})\n                logits = outs[ONNX_OUT_MAP[\"label\"]].astype(np.float32)\n                emb    = outs[ONNX_OUT_MAP[\"embedding\"]].astype(np.float32)\n            else:\n                out    = infer_fn(inputs=tf.convert_to_tensor(x))\n                logits = out[\"label\"].numpy().astype(np.float32)\n                emb    = out[\"embedding\"].numpy().astype(np.float32)\n            scores[br:wr, MAPPED_POS] = logits[:, MAPPED_BC_IDX]\n            embs  [br:wr]             = emb\n            for pos_idx, bc_idxs in proxy_map.items():\n                bc_arr = np.array(bc_idxs, dtype=np.int32)\n                scores[br:wr, pos_idx] = logits[:, bc_arr].max(axis=1)\n            del x, logits, emb, batch_audio\n            gc.collect()\n    meta_df = pd.DataFrame({\"row_id\": row_ids, \"filename\": filenames,\n                             \"site\": sites, \"hour_utc\": hours})\n    return meta_df, scores, embs\nprint(\"Perch inference engine defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"USE_ONNX = {USE_ONNX}\")\nEXTERNAL_CACHE_DIRS = [\n    Path(\"/kaggle/input/notebooks/vyankteshdwivedi/notebook1b25083f0d\"),\n    Path(\"/kaggle/input/datasets/jaejohn/perch-meta\"),\n]\nCACHE_META_LOCAL = WORK_DIR / \"perch_meta.parquet\"\nCACHE_NPZ_LOCAL  = WORK_DIR / \"perch_arrays.npz\"\ndef _find_external_cache():\n    for d in EXTERNAL_CACHE_DIRS:\n        meta = d / \"perch_meta.parquet\"\n        npz  = d / \"perch_arrays.npz\"\n        if meta.exists() and npz.exists():\n            return meta, npz\n    return None, None\nSCORE_KEYS = [\"scores\", \"sc\", \"logits\", \"perch_scores\", \"preds\", \"arr_0\"]\nEMB_KEYS   = [\"embs\", \"emb\", \"embeddings\", \"features\", \"perch_embs\", \"arr_1\"]\ndef _pick_array(arr, candidates, shape_hint_cols):\n    for k in candidates:\n        if k in arr.files:\n            return arr[k], k\n    for k in arr.files:\n        v = arr[k]\n        if v.ndim == 2 and v.shape[1] == shape_hint_cols:\n            return v, k\n    raise KeyError(f\"None of {candidates} found in npz. Available keys: {arr.files}\")\ndef _build_cache():\n    print(f\"Building Perch cache from {len(full_files)} training files...\")\n    train_paths = [BASE / \"train_soundscapes\" / fn for fn in full_files]\n    train_paths = [p for p in train_paths if p.exists()]\n    t0 = time.time()\n    meta_built, sc_built, emb_built = run_perch(train_paths, batch_files=CFG[\"batch_files\"], verbose=True)\n    print(f\"  Perch done in {time.time()-t0:.1f}s  scores={sc_built.shape} embs={emb_built.shape}\")\n    meta_built.to_parquet(CACHE_META_LOCAL)\n    np.savez(CACHE_NPZ_LOCAL, scores=sc_built.astype(np.float32),\n             embs=emb_built.astype(np.float32), primary_labels=np.array(PRIMARY_LABELS))\n    return CACHE_META_LOCAL, CACHE_NPZ_LOCAL\next_meta, ext_npz = _find_external_cache()\nif ext_meta is not None:\n    CACHE_META, CACHE_NPZ = ext_meta, ext_npz\n    print(f\"Using external cache: {CACHE_META.parent}\")\nelif CACHE_META_LOCAL.exists() and CACHE_NPZ_LOCAL.exists():\n    CACHE_META, CACHE_NPZ = CACHE_META_LOCAL, CACHE_NPZ_LOCAL\n    print(f\"Using local cache: {WORK_DIR}\")\nelse:\n    print(\"No cache found — building from scratch (~2.5 min)\")\n    CACHE_META, CACHE_NPZ = _build_cache()\nprint(\"Loading Perch cache from:\", CACHE_META.parent)\nmeta_tr = pd.read_parquet(CACHE_META)\n_arr    = np.load(CACHE_NPZ)\nsc_tr_raw,  sk = _pick_array(_arr, SCORE_KEYS, N_CLASSES)\nemb_tr_raw, ek = _pick_array(_arr, EMB_KEYS,   1536)\nprint(f\"  scores <- '{sk}'  shape={sc_tr_raw.shape}\")\nprint(f\"  embs   <- '{ek}'  shape={emb_tr_raw.shape}\")\nsc_tr  = sc_tr_raw.astype(np.float32)\nemb_tr = emb_tr_raw.astype(np.float32)\nif \"primary_labels\" in _arr.files:\n    if _arr[\"primary_labels\"].tolist() != PRIMARY_LABELS:\n        print(\"  WARNING: cached primary_labels differ!\")\n    else:\n        print(\"  primary_labels schema OK\")\nif \"row_id\" not in meta_tr.columns:\n    print(\"  row_id missing — reconstructing\")\n    if \"end_sec\" in meta_tr.columns:\n        end_sec = meta_tr[\"end_sec\"].astype(int)\n    elif \"window_idx\" in meta_tr.columns:\n        end_sec = (meta_tr[\"window_idx\"].astype(int) + 1) * 5\n    else:\n        end_sec = np.tile(np.arange(5, 65, 5), len(meta_tr) // N_WINDOWS)\n    meta_tr[\"row_id\"] = (\n        meta_tr[\"filename\"].str.replace(\".ogg\", \"\", regex=False)\n        + \"_\" + end_sec.astype(str)\n    )\nrow_id_to_index = full_rows.set_index(\"row_id\")[\"index\"]\nmissing_rows = set(meta_tr[\"row_id\"]) - set(row_id_to_index.index)\nif missing_rows:\n    raise RuntimeError(f\"Cache has {len(missing_rows)} row_ids not in labeled set.\")\nY_FULL_aligned = Y_SC[row_id_to_index.loc[meta_tr[\"row_id\"]].to_numpy()]\nprint(f\"sc_tr: {sc_tr.shape}  emb_tr: {emb_tr.shape}  Y_FULL_aligned: {Y_FULL_aligned.shape}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def macro_auc(y_true, y_score):\n    keep = y_true.sum(axis=0) > 0\n    return roc_auc_score(y_true[:, keep], y_score[:, keep], average=\"macro\")\n\ndef honest_oof_auc(scores, Y, meta_df, n_splits=5, label=\"scores\"):\n    groups = meta_df[\"filename\"].to_numpy()\n    gkf    = GroupKFold(n_splits=n_splits)\n    oof    = np.zeros_like(scores, dtype=np.float32)\n    for fold, (tr_idx, va_idx) in enumerate(gkf.split(scores, groups=groups), 1):\n        oof[va_idx] = scores[va_idx]\n    auc = macro_auc(Y, oof)\n    print(f\"[{label}] honest OOF macro-AUC: {auc:.6f}\")\n    return auc, oof","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def smooth_predictions(probs, n_windows=12, alpha=0.3):\n    N, C = probs.shape\n    assert N % n_windows == 0\n    view = probs.reshape(-1, n_windows, C).copy()\n    prev_w = np.concatenate([view[:, :1, :],  view[:, :-1, :]], axis=1)\n    next_w = np.concatenate([view[:, 1:,  :], view[:, -1:, :]], axis=1)\n    smoothed = (1 - alpha) * view + 0.5 * alpha * (prev_w + next_w)\n    return smoothed.reshape(N, C)\nprint(\"Temporal smoothing helper defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_prior_tables(sc_df, Y_labels):\n    sc_df = sc_df.reset_index(drop=True)\n    global_p = Y_labels.mean(axis=0).astype(np.float32)\n    site_keys = sorted(sc_df[\"site\"].dropna().astype(str).unique())\n    site_to_i = {k: i for i, k in enumerate(site_keys)}\n    site_p = np.zeros((len(site_keys), Y_labels.shape[1]), dtype=np.float32)\n    site_n = np.zeros(len(site_keys), dtype=np.float32)\n    for s in site_keys:\n        i = site_to_i[s]; mask = sc_df[\"site\"].astype(str).values == s\n        site_n[i] = mask.sum(); site_p[i] = Y_labels[mask].mean(axis=0)\n    hour_keys = sorted(sc_df[\"hour_utc\"].dropna().astype(int).unique())\n    hour_to_i = {h: i for i, h in enumerate(hour_keys)}\n    hour_p = np.zeros((len(hour_keys), Y_labels.shape[1]), dtype=np.float32)\n    hour_n = np.zeros(len(hour_keys), dtype=np.float32)\n    for h in hour_keys:\n        i = hour_to_i[h]; mask = sc_df[\"hour_utc\"].astype(int).values == h\n        hour_n[i] = mask.sum(); hour_p[i] = Y_labels[mask].mean(axis=0)\n    sh_keys = sorted({(str(s), int(h)) for s, h in zip(sc_df[\"site\"].dropna(), sc_df[\"hour_utc\"].dropna())\n                      if not pd.isna(s) and not pd.isna(h)})\n    sh_to_i = {k: i for i, k in enumerate(sh_keys)}\n    sh_p = np.zeros((len(sh_keys), Y_labels.shape[1]), dtype=np.float32)\n    sh_n = np.zeros(len(sh_keys), dtype=np.float32)\n    for (s, h) in sh_keys:\n        i = sh_to_i[(s, h)]\n        mask = (sc_df[\"site\"].astype(str).values == s) & (sc_df[\"hour_utc\"].astype(int).values == h)\n        sh_n[i] = mask.sum(); sh_p[i] = Y_labels[mask].mean(axis=0)\n    return {\"global_p\": global_p,\n            \"site_to_i\": site_to_i, \"site_p\": site_p, \"site_n\": site_n,\n            \"hour_to_i\": hour_to_i, \"hour_p\": hour_p, \"hour_n\": hour_n,\n            \"sh_to_i\": sh_to_i, \"sh_p\": sh_p, \"sh_n\": sh_n}\n\ndef apply_prior(scores, sites, hours, tables, lambda_prior=0.4):\n    eps = 1e-4; n = len(scores); out = scores.copy()\n    p = np.tile(tables[\"global_p\"], (n, 1))\n    for i, h in enumerate(hours):\n        h = int(h)\n        if h in tables[\"hour_to_i\"]:\n            j = tables[\"hour_to_i\"][h]; nh = tables[\"hour_n\"][j]; w = nh / (nh + 8.0)\n            p[i] = w * tables[\"hour_p\"][j] + (1 - w) * tables[\"global_p\"]\n    for i, s in enumerate(sites):\n        s = str(s)\n        if s in tables[\"site_to_i\"]:\n            j = tables[\"site_to_i\"][s]; ns = tables[\"site_n\"][j]; w = ns / (ns + 8.0)\n            p[i] = w * tables[\"site_p\"][j] + (1 - w) * p[i]\n    if \"sh_to_i\" in tables:\n        for i, (s, h) in enumerate(zip(sites, hours)):\n            key = (str(s), int(h))\n            if key in tables[\"sh_to_i\"]:\n                j = tables[\"sh_to_i\"][key]; nsh = tables[\"sh_n\"][j]; w = nsh / (nsh + 4.0)\n                p[i] = w * tables[\"sh_p\"][j] + (1 - w) * p[i]\n    p = np.clip(p, eps, 1 - eps)\n    out += lambda_prior * (np.log(p) - np.log1p(-p))\n    return out.astype(np.float32)\nprint(\"Prior tables defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def file_confidence_scale(probs, n_windows=12, top_k=2, power=0.4):\n    N, C = probs.shape\n    assert N % n_windows == 0\n    view       = probs.reshape(-1, n_windows, C)\n    sorted_v   = np.sort(view, axis=1)\n    top_k_mean = sorted_v[:, -top_k:, :].mean(axis=1, keepdims=True)\n    scale  = np.power(top_k_mean, power)\n    scaled = view * scale\n    return scaled.reshape(N, C)\nprint(\"File-level confidence scaling defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CLASS_NAME_MAP = taxonomy.set_index(\"primary_label\")[\"class_name\"].to_dict()\nTEXTURE_TAXA   = {\"Amphibia\", \"Insecta\"}\ntemperatures = np.ones(N_CLASSES, dtype=np.float32)\nfor ci, label in enumerate(PRIMARY_LABELS):\n    cls = CLASS_NAME_MAP.get(label, \"Aves\")\n    if cls in TEXTURE_TAXA:\n        temperatures[ci] = 0.95\n    else:\n        temperatures[ci] = 1.10\nn_texture = (temperatures < 1.0).sum()\nn_event   = (temperatures > 1.0).sum()\nprint(f\"Temperatures: {n_event} event species (T=1.10), {n_texture} texture species (T=0.95)\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.neural_network import MLPClassifier\ndef build_class_freq_weights(Y, cap=10.0):\n    total     = Y.shape[0]\n    pos_count = Y.sum(axis=0).astype(np.float32) + 1.0\n    freq      = pos_count / total\n    weights   = 1.0 / (freq ** 0.5)\n    weights   = np.clip(weights, 1.0, cap)\n    weights   = weights / weights.mean()\n    return weights.astype(np.float32)\ndef build_sequential_features(scores_col, n_windows=12):\n    N = len(scores_col)\n    assert N % n_windows == 0\n    x     = scores_col.reshape(-1, n_windows)\n    prev  = np.concatenate([x[:, :1], x[:, :-1]], axis=1)\n    next_ = np.concatenate([x[:, 1:], x[:, -1:]], axis=1)\n    mean  = np.repeat(x.mean(axis=1), n_windows)\n    max_  = np.repeat(x.max(axis=1),  n_windows)\n    std   = np.repeat(x.std(axis=1),  n_windows)\n    return prev.reshape(-1), next_.reshape(-1), mean, max_, std\ndef train_mlp_probes(emb, scores_raw, Y, min_pos=5, pca_dim=64, alpha_blend=0.4):\n    scaler = StandardScaler()\n    emb_s  = scaler.fit_transform(emb)\n    pca    = PCA(n_components=min(pca_dim, emb_s.shape[1] - 1))\n    Z      = pca.fit_transform(emb_s).astype(np.float32)\n    print(f\"Embedding: {emb.shape} -> PCA: {Z.shape}  \"\n          f\"(variance retained: {pca.explained_variance_ratio_.sum():.2%})\")\n    class_weights = build_class_freq_weights(Y, cap=10.0)\n    probe_models = {}\n    active = np.where(Y.sum(axis=0) >= min_pos)[0]\n    print(f\"Training MLP probes for {len(active)} species (>= {min_pos} pos windows)...\")\n    MAX_ROWS = 3000\n    for ci in tqdm(active, desc=\"MLP probes\"):\n        y = Y[:, ci]\n        if y.sum() == 0 or y.sum() == len(y):\n            continue\n        prev, next_, mean, max_, std = build_sequential_features(scores_raw[:, ci])\n        X = np.hstack([Z, scores_raw[:, ci:ci+1],\n                       prev[:, None], next_[:, None],\n                       mean[:, None], max_[:, None], std[:, None]])\n        n_pos = int(y.sum()); n_neg = len(y) - n_pos\n        pos_idx = np.where(y == 1)[0]\n        w      = float(class_weights[ci])\n        repeat = max(1, int(round(w * n_neg / max(n_pos, 1))))\n        repeat = min(repeat, 8)\n        if n_pos * repeat + len(y) > MAX_ROWS:\n            repeat = max(1, (MAX_ROWS - len(y)) // max(n_pos, 1))\n        X_bal = np.vstack([X, np.tile(X[pos_idx], (repeat, 1))])\n        y_bal = np.concatenate([y, np.ones(n_pos * repeat, dtype=y.dtype)])\n        clf = MLPClassifier(\n            hidden_layer_sizes=CFG[\"mlp_params\"][\"hidden_layer_sizes\"],\n            activation=\"relu\",\n            max_iter=CFG[\"mlp_params\"][\"max_iter\"],\n            early_stopping=True,\n            validation_fraction=0.15,\n            n_iter_no_change=CFG[\"mlp_params\"][\"n_iter_no_change\"],\n            random_state=42,\n            learning_rate_init=5e-4,\n            alpha=0.005,\n        )\n        clf.fit(X_bal, y_bal)\n        probe_models[ci] = clf\n    print(f\"Trained {len(probe_models)} MLP probes\")\n    return probe_models, scaler, pca, alpha_blend\ndef apply_mlp_probes(emb_test, scores_test, probe_models, scaler, pca, alpha_blend=0.4):\n    emb_s  = scaler.transform(emb_test)\n    Z_test = pca.transform(emb_s).astype(np.float32)\n    result = scores_test.copy()\n    for ci, clf in probe_models.items():\n        prev, next_, mean, max_, std = build_sequential_features(scores_test[:, ci])\n        X_test = np.hstack([Z_test, scores_test[:, ci:ci+1],\n                             prev[:, None], next_[:, None],\n                             mean[:, None], max_[:, None], std[:, None]])\n        prob  = clf.predict_proba(X_test)[:, 1].astype(np.float32)\n        logit = np.log(prob + 1e-7) - np.log(1 - prob + 1e-7)\n        result[:, ci] = (1 - alpha_blend) * scores_test[:, ci] + alpha_blend * logit\n    return result\nprint(f\"MLP probes ready — hidden={CFG['mlp_params']['hidden_layer_sizes']} pca={CFG['mlp_pca_dim']}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nclass VectorizedMLPProbes(nn.Module):\n    def __init__(self, probe_models):\n        super().__init__()\n        self.valid_classes = sorted(probe_models.keys())\n        V = len(self.valid_classes)\n        if V == 0:\n            self.weights = nn.ParameterList()\n            self.biases  = nn.ParameterList()\n            self.n_layers = 0\n            return\n        sample = probe_models[self.valid_classes[0]]\n        self.n_layers = len(sample.coefs_)\n        self.weights  = nn.ParameterList()\n        self.biases   = nn.ParameterList()\n        for layer_idx in range(self.n_layers):\n            W = np.stack([probe_models[c].coefs_[layer_idx] for c in self.valid_classes], axis=0)\n            b = np.stack([probe_models[c].intercepts_[layer_idx] for c in self.valid_classes], axis=0)\n            self.weights.append(nn.Parameter(torch.tensor(W, dtype=torch.float32), requires_grad=False))\n            self.biases.append(nn.Parameter(torch.tensor(b, dtype=torch.float32), requires_grad=False))\n    def forward(self, x):\n        h = x\n        for i in range(self.n_layers):\n            h = torch.bmm(h, self.weights[i]) + self.biases[i].unsqueeze(1)\n            if i < self.n_layers - 1:\n                h = torch.relu(h)\n        return h.squeeze(-1)\ndef apply_mlp_probes_vectorized(emb_test, scores_test, probe_models, scaler, pca, alpha_blend=0.4):\n    if len(probe_models) == 0:\n        return scores_test.copy()\n    emb_s  = scaler.transform(emb_test)\n    Z_test = pca.transform(emb_s).astype(np.float32)\n    valid_classes = sorted(probe_models.keys())\n    V = len(valid_classes); N = len(scores_test)\n    raw  = scores_test[:, valid_classes].T\n    n_files = N // N_WINDOWS\n    raw_view = raw.reshape(V, n_files, N_WINDOWS)\n    prev = np.concatenate([raw_view[:, :, :1], raw_view[:, :, :-1]], axis=2).reshape(V, N)\n    nxt  = np.concatenate([raw_view[:, :, 1:], raw_view[:, :, -1:]], axis=2).reshape(V, N)\n    mean = np.repeat(raw_view.mean(axis=2), N_WINDOWS, axis=1)\n    mx   = np.repeat(raw_view.max(axis=2),  N_WINDOWS, axis=1)\n    std  = np.repeat(raw_view.std(axis=2),  N_WINDOWS, axis=1)\n    scalar_feats = np.stack([raw, prev, nxt, mean, mx, std], axis=-1).astype(np.float32)\n    Z_expanded = np.broadcast_to(Z_test, (V, N, Z_test.shape[1]))\n    X_all = np.concatenate([Z_expanded.astype(np.float32), scalar_feats], axis=-1)\n    vec_probe = VectorizedMLPProbes(probe_models)\n    vec_probe.eval()\n    with torch.no_grad():\n        preds = vec_probe(torch.tensor(X_all)).numpy()\n    result = scores_test.copy()\n    base_valid = scores_test[:, valid_classes]\n    result[:, valid_classes] = (1.0 - alpha_blend) * base_valid + alpha_blend * preds.T\n    return result\nprint(\"Vectorized MLP probe inference defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.isotonic import IsotonicRegression\ndef calibrate_and_optimize_thresholds(oof_probs, Y_FULL, threshold_grid=None, n_windows=12):\n    if threshold_grid is None:\n        threshold_grid = [0.25, 0.30, 0.35, 0.40, 0.45, 0.50, 0.55, 0.60, 0.65, 0.70]\n    n_samples, n_cls = oof_probs.shape\n    thresholds = np.full(n_cls, 0.5, dtype=np.float32)\n    n_files    = n_samples // n_windows\n    file_oof   = oof_probs.reshape(n_files, n_windows, n_cls).max(axis=1)\n    file_y     = Y_FULL.reshape(n_files, n_windows, n_cls).max(axis=1)\n    n_calibrated = 0\n    for c in range(n_cls):\n        y_true = file_y[:, c]; y_prob = file_oof[:, c]\n        if y_true.sum() < 3: continue\n        try:\n            ir = IsotonicRegression(out_of_bounds=\"clip\")\n            ir.fit(y_prob, y_true); y_cal = ir.transform(y_prob)\n        except Exception: y_cal = y_prob\n        best_f1, best_t = 0.0, 0.5\n        for t in threshold_grid:\n            pred = (y_cal >= t).astype(int)\n            tp = ((pred==1) & (y_true==1)).sum(); fp = ((pred==1) & (y_true==0)).sum(); fn = ((pred==0) & (y_true==1)).sum()\n            prec = tp / (tp + fp + 1e-8); rec = tp / (tp + fn + 1e-8)\n            f1 = 2 * prec * rec / (prec + rec + 1e-8)\n            if f1 > best_f1: best_f1, best_t = f1, t\n        thresholds[c] = best_t; n_calibrated += 1\n    print(f\"Calibrated {n_calibrated} classes\")\n    print(f\"Mean threshold: {thresholds.mean():.3f}\")\n    print(f\"Range: [{thresholds.min():.2f}, {thresholds.max():.2f}]\")\n    return thresholds\ndef apply_per_class_thresholds(scores, thresholds):\n    C = scores.shape[1]; assert C == len(thresholds)\n    scaled = np.copy(scores)\n    for c in range(C):\n        t = thresholds[c]; above = scores[:, c] > t\n        scaled[ above, c] = 0.5 + 0.5 * (scores[ above, c] - t) / (1 - t + 1e-8)\n        scaled[~above, c] = 0.5 * scores[~above, c] / (t + 1e-8)\n    return np.clip(scaled, 0.0, 1.0)\nprint(\"Isotonic calibration + per-class threshold optimization defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rank_aware_scaling(probs, n_windows=12, power=0.4):\n    N, C = probs.shape\n    assert N % n_windows == 0\n    view     = probs.reshape(-1, n_windows, C)\n    file_max = view.max(axis=1, keepdims=True)\n    scale  = np.power(file_max, power)\n    scaled = view * scale\n    return scaled.reshape(N, C)\nprint(\"Rank-aware scaling defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adaptive_delta_smooth(probs, n_windows=12, base_alpha=0.20):\n    N, C = probs.shape\n    assert N % n_windows == 0\n    result = probs.copy()\n    view   = probs.reshape(-1, n_windows, C)\n    out    = result.reshape(-1, n_windows, C)\n    for t in range(n_windows):\n        conf = view[:, t, :].max(axis=-1, keepdims=True)\n        alpha = base_alpha * (1.0 - conf)\n        if t == 0:\n            neighbor_avg = (view[:, t, :] + view[:, t+1, :]) / 2.0\n        elif t == n_windows - 1:\n            neighbor_avg = (view[:, t-1, :] + view[:, t, :]) / 2.0\n        else:\n            neighbor_avg = (view[:, t-1, :] + view[:, t+1, :]) / 2.0\n        out[:, t, :] = (1.0 - alpha) * view[:, t, :] + alpha * neighbor_avg\n    return result\nprint(\"Adaptive delta smoothing defined\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nclass SelectiveSSM(nn.Module):\n    def __init__(self, d_model, d_state=16, d_conv=4):\n        super().__init__()\n        self.d_model = d_model; self.d_state = d_state\n        self.in_proj = nn.Linear(d_model, 2 * d_model, bias=False)\n        self.conv1d = nn.Conv1d(d_model, d_model, d_conv, padding=d_conv - 1, groups=d_model)\n        self.dt_proj = nn.Linear(d_model, d_model, bias=True)\n        A = torch.arange(1, d_state + 1, dtype=torch.float32).unsqueeze(0).expand(d_model, -1)\n        self.A_log = nn.Parameter(torch.log(A))\n        self.D = nn.Parameter(torch.ones(d_model))\n        self.B_proj = nn.Linear(d_model, d_state, bias=False)\n        self.C_proj = nn.Linear(d_model, d_state, bias=False)\n        self.out_proj = nn.Linear(d_model, d_model, bias=False)\n    def forward(self, x):\n        B_sz, T, D = x.shape\n        xz = self.in_proj(x); x_ssm, z = xz.chunk(2, dim=-1)\n        x_conv = self.conv1d(x_ssm.transpose(1, 2))[:, :, :T].transpose(1, 2)\n        x_conv = F.silu(x_conv)\n        dt = F.softplus(self.dt_proj(x_conv))\n        A = -torch.exp(self.A_log); B = self.B_proj(x_conv); C = self.C_proj(x_conv)\n        h = torch.zeros(B_sz, D, self.d_state, device=x.device)\n        ys = []\n        for t in range(T):\n            dA = torch.exp(A[None] * dt[:, t, :, None])\n            dB = dt[:, t, :, None] * B[:, t, None, :]\n            h = h * dA + x[:, t, :, None] * dB\n            ys.append((h * C[:, t, None, :]).sum(-1))\n        y = torch.stack(ys, dim=1)\n        return y + x * self.D[None, None, :]\nclass LightProtoSSM(nn.Module):\n    def __init__(self, d_input=1536, d_model=128, d_state=16, n_classes=234, n_windows=12,\n                 dropout=0.15, n_sites=20, meta_dim=16, use_cross_attn=True, cross_attn_heads=2):\n        super().__init__()\n        self.n_classes = n_classes; self.n_windows = n_windows; self.use_cross_attn = use_cross_attn\n        self.input_proj = nn.Sequential(nn.Linear(d_input, d_model), nn.LayerNorm(d_model), nn.GELU(), nn.Dropout(dropout))\n        self.pos_enc  = nn.Parameter(torch.randn(1, n_windows, d_model) * 0.02)\n        self.site_emb = nn.Embedding(n_sites, meta_dim); self.hour_emb = nn.Embedding(24, meta_dim)\n        self.meta_proj = nn.Linear(2 * meta_dim, d_model)\n        self.ssm_fwd  = nn.ModuleList([SelectiveSSM(d_model, d_state) for _ in range(2)])\n        self.ssm_bwd  = nn.ModuleList([SelectiveSSM(d_model, d_state) for _ in range(2)])\n        self.ssm_merge= nn.ModuleList([nn.Linear(2 * d_model, d_model) for _ in range(2)])\n        self.ssm_norm = nn.ModuleList([nn.LayerNorm(d_model) for _ in range(2)])\n        self.drop     = nn.Dropout(dropout)\n        if use_cross_attn:\n            self.cross_attn = nn.ModuleList([\n                nn.MultiheadAttention(d_model, num_heads=cross_attn_heads, dropout=dropout, batch_first=True)\n                for _ in range(2)])\n            self.cross_norm = nn.ModuleList([nn.LayerNorm(d_model) for _ in range(2)])\n        self.prototypes   = nn.Parameter(torch.randn(n_classes, d_model) * 0.02)\n        self.proto_temp   = nn.Parameter(torch.tensor(5.0))\n        self.class_bias   = nn.Parameter(torch.zeros(n_classes))\n        self.fusion_alpha = nn.Parameter(torch.zeros(n_classes))\n    def init_prototypes(self, emb_tensor, labels_tensor):\n        with torch.no_grad():\n            h = self.input_proj(emb_tensor)\n            for c in range(self.n_classes):\n                mask = labels_tensor[:, c] > 0.5\n                if mask.sum() > 0:\n                    self.prototypes.data[c] = F.normalize(h[mask].mean(0), dim=0)\n    def forward(self, emb, perch_logits=None, site_ids=None, hours=None):\n        B, T, _ = emb.shape\n        h = self.input_proj(emb) + self.pos_enc[:, :T, :]\n        if site_ids is not None and hours is not None:\n            meta = self.meta_proj(torch.cat([self.site_emb(site_ids), self.hour_emb(hours)], dim=-1))\n            h = h + meta[:, None, :]\n        for i, (fwd, bwd, merge, norm) in enumerate(zip(self.ssm_fwd, self.ssm_bwd, self.ssm_merge, self.ssm_norm)):\n            res = h; h_f = fwd(h); h_b = bwd(h.flip(1)).flip(1)\n            h   = self.drop(merge(torch.cat([h_f, h_b], dim=-1))); h = norm(h + res)\n            if self.use_cross_attn:\n                attn_out, _ = self.cross_attn[i](h, h, h); h = self.cross_norm[i](h + attn_out)\n        h_n = F.normalize(h, dim=-1); p_n = F.normalize(self.prototypes, dim=-1)\n        sim = (torch.matmul(h_n, p_n.T) * F.softplus(self.proto_temp) + self.class_bias[None, None, :])\n        if perch_logits is not None:\n            alpha = torch.sigmoid(self.fusion_alpha)[None, None, :]\n            out   = alpha * sim + (1 - alpha) * perch_logits\n        else:\n            out = sim\n        return out\ndef train_light_proto_ssm(emb_full, scores_full, Y_full, meta_full, n_epochs=40, patience=8,\n                           lr=1e-3, n_sites=20, d_model=128, verbose=False):\n    n_files = len(emb_full) // N_WINDOWS\n    emb_f   = emb_full.reshape(n_files, N_WINDOWS, -1)\n    log_f   = scores_full.reshape(n_files, N_WINDOWS, -1)\n    lab_f   = Y_full.reshape(n_files, N_WINDOWS, -1).astype(np.float32)\n    fnames  = meta_full[\"filename\"].unique()\n    sites_u = sorted(meta_full[\"site\"].unique()); site2i = {s: i + 1 for i, s in enumerate(sites_u)}\n    site_ids = np.array([min(site2i.get(meta_full.loc[meta_full[\"filename\"]==fn,\"site\"].iloc[0], 0), n_sites-1) for fn in fnames], dtype=np.int64)\n    hour_ids = np.array([int(meta_full.loc[meta_full[\"filename\"]==fn,\"hour_utc\"].iloc[0]) % 24 for fn in fnames], dtype=np.int64)\n    model = LightProtoSSM(n_classes=N_CLASSES, n_sites=n_sites, use_cross_attn=True, cross_attn_heads=2, d_model=d_model)\n    model.init_prototypes(torch.tensor(emb_full, dtype=torch.float32), torch.tensor(Y_full, dtype=torch.float32))\n    print(f\"LightProtoSSM params: {sum(p.numel() for p in model.parameters() if p.requires_grad):,}\")\n    emb_t  = torch.tensor(emb_f,    dtype=torch.float32); log_t = torch.tensor(log_f, dtype=torch.float32)\n    lab_t  = torch.tensor(lab_f,    dtype=torch.float32); site_t = torch.tensor(site_ids, dtype=torch.long)\n    hour_t = torch.tensor(hour_ids, dtype=torch.long)\n    pos_cnt    = lab_t.sum(dim=(0, 1)); total = lab_t.shape[0] * lab_t.shape[1]\n    pos_weight = ((total - pos_cnt) / (pos_cnt + 1)).clamp(max=25.0)\n    opt   = torch.optim.AdamW(model.parameters(), lr=lr, weight_decay=1e-3)\n    sched = torch.optim.lr_scheduler.OneCycleLR(opt, max_lr=lr, epochs=n_epochs, steps_per_epoch=1, pct_start=0.1, anneal_strategy=\"cos\")\n    best_loss, best_state, wait = float(\"inf\"), None, 0\n    swa_model = torch.optim.swa_utils.AveragedModel(model)\n    swa_start = int(n_epochs * 0.65); swa_sched = torch.optim.swa_utils.SWALR(opt, swa_lr=4e-4)\n    for ep in range(n_epochs):\n        model.train()\n        out  = model(emb_t, log_t, site_ids=site_t, hours=hour_t)\n        loss = F.binary_cross_entropy_with_logits(out, lab_t, pos_weight=pos_weight[None, None, :]) + 0.15 * F.mse_loss(out, log_t)\n        opt.zero_grad(); loss.backward(); torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0); opt.step()\n        if ep >= swa_start: swa_model.update_parameters(model); swa_sched.step()\n        else: sched.step()\n        if loss.item() < best_loss: best_loss = loss.item(); best_state = {k: v.clone() for k, v in model.state_dict().items()}; wait = 0\n        else: wait += 1\n        if wait >= patience: break\n    if ep >= swa_start: torch.optim.swa_utils.update_bn(emb_t.unsqueeze(0), swa_model); model = swa_model\n    else: model.load_state_dict(best_state)\n    model.eval()\n    return model, site2i\ndef run_tta_proto(proto_model, emb_files, sc_files, site_t, hour_t, shifts=[0, 1, -1, 2, -2]):\n    proto_model.eval(); all_preds = []\n    emb_t = torch.tensor(emb_files, dtype=torch.float32); sc_t = torch.tensor(sc_files, dtype=torch.float32)\n    for shift in shifts:\n        if shift == 0: e_shifted = emb_t; s_shifted = sc_t\n        else: e_shifted = torch.roll(emb_t, shift, dims=1); s_shifted = torch.roll(sc_t, shift, dims=1)\n        with torch.no_grad():\n            out = proto_model(e_shifted, s_shifted, site_ids=site_t, hours=hour_t).numpy()\n        if shift != 0: out = np.roll(out, -shift, axis=1)\n        all_preds.append(out)\n    return np.mean(all_preds, axis=0)\nclass ResidualSSM(nn.Module):\n    def __init__(self, d_input=1536, d_scores=234, d_model=64, d_state=8, n_classes=234,\n                 n_windows=12, dropout=0.1, n_sites=20, meta_dim=8):\n        super().__init__()\n        self.n_classes = n_classes\n        self.input_proj = nn.Sequential(nn.Linear(d_input + d_scores, d_model), nn.LayerNorm(d_model), nn.GELU(), nn.Dropout(dropout))\n        self.site_emb  = nn.Embedding(n_sites, meta_dim); self.hour_emb = nn.Embedding(24, meta_dim)\n        self.meta_proj = nn.Linear(2 * meta_dim, d_model)\n        self.pos_enc   = nn.Parameter(torch.randn(1, n_windows, d_model) * 0.02)\n        self.ssm_fwd   = SelectiveSSM(d_model, d_state); self.ssm_bwd = SelectiveSSM(d_model, d_state)\n        self.ssm_merge = nn.Linear(2 * d_model, d_model); self.ssm_norm = nn.LayerNorm(d_model); self.ssm_drop = nn.Dropout(dropout)\n        self.output_head = nn.Linear(d_model, n_classes)\n        nn.init.zeros_(self.output_head.weight); nn.init.zeros_(self.output_head.bias)\n    def forward(self, emb, first_pass, site_ids=None, hours=None):\n        B, T, _ = emb.shape\n        x = torch.cat([emb, first_pass], dim=-1); h = self.input_proj(x) + self.pos_enc[:, :T, :]\n        if site_ids is not None and hours is not None:\n            meta = self.meta_proj(torch.cat([self.site_emb(site_ids.clamp(0, self.site_emb.num_embeddings-1)),\n                                              self.hour_emb(hours.clamp(0, 23))], dim=-1))\n            h = h + meta.unsqueeze(1)\n        res = h; h_f = self.ssm_fwd(h); h_b = self.ssm_bwd(h.flip(1)).flip(1)\n        h   = self.ssm_drop(self.ssm_merge(torch.cat([h_f, h_b], dim=-1))); h = self.ssm_norm(h + res)\n        return self.output_head(h)\ndef train_residual_ssm(emb_full, first_pass_flat, Y_full, site_ids, hour_ids,\n                        n_epochs=30, patience=8, lr=1e-3, correction_weight=0.30, verbose=False):\n    n_files    = len(emb_full) // N_WINDOWS\n    emb_f      = emb_full.reshape(n_files, N_WINDOWS, -1)\n    fp_f       = first_pass_flat.reshape(n_files, N_WINDOWS, -1)\n    lab_f      = Y_full.reshape(n_files, N_WINDOWS, -1).astype(np.float32)\n    fp_prob    = 1.0 / (1.0 + np.exp(-np.clip(fp_f, -30, 30)))\n    residuals  = lab_f - fp_prob\n    n_val = max(1, int(n_files * 0.15)); rng = torch.Generator(); rng.manual_seed(42)\n    perm = torch.randperm(n_files, generator=rng).numpy()\n    val_i = perm[:n_val]; train_i = perm[n_val:]\n    emb_t = torch.tensor(emb_f, dtype=torch.float32); fp_t = torch.tensor(fp_f, dtype=torch.float32)\n    res_t = torch.tensor(residuals, dtype=torch.float32)\n    site_t = torch.tensor(site_ids, dtype=torch.long); hour_t = torch.tensor(hour_ids, dtype=torch.long)\n    model = ResidualSSM(n_classes=N_CLASSES)\n    opt   = torch.optim.AdamW(model.parameters(), lr=lr, weight_decay=1e-3)\n    sched = torch.optim.lr_scheduler.OneCycleLR(opt, max_lr=lr, epochs=n_epochs, steps_per_epoch=1, pct_start=0.1, anneal_strategy=\"cos\")\n    best_loss, best_state, wait = float(\"inf\"), None, 0\n    for ep in range(n_epochs):\n        model.train()\n        corr = model(emb_t[train_i], fp_t[train_i], site_ids=site_t[train_i], hours=hour_t[train_i])\n        loss = F.mse_loss(corr, res_t[train_i])\n        opt.zero_grad(); loss.backward(); torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0); opt.step(); sched.step()\n        model.eval()\n        with torch.no_grad():\n            val_corr = model(emb_t[val_i], fp_t[val_i], site_ids=site_t[val_i], hours=hour_t[val_i])\n            val_loss = F.mse_loss(val_corr, res_t[val_i])\n        if val_loss.item() < best_loss: best_loss = val_loss.item(); best_state = {k: v.clone() for k, v in model.state_dict().items()}; wait = 0\n        else: wait += 1\n        if wait >= patience: break\n    model.load_state_dict(best_state)\n    return model, correction_weight\nprint(\"Sequence Models Initialized\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"baseline_auc = None\noof_raw      = None\nif CFG[\"run_oof\"]:\n    print(\"Running honest OOF evaluation on training data...\")\n    baseline_auc, oof_raw = honest_oof_auc(sc_tr, Y_FULL_aligned, meta_tr,\n                                             n_splits=CFG[\"oof_n_splits\"], label=\"raw Perch\")\n    print(f\"Baseline OOF AUC: {baseline_auc:.6f}\")\nelse:\n    print(\"Submit mode: skipping OOF evaluation\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_paths = sorted((BASE / \"test_soundscapes\").glob(\"*.ogg\"))\nIS_DRY_RUN = len(test_paths) == 0\nif IS_DRY_RUN:\n    n = CFG[\"dryrun_n_files\"] or 20\n    print(f\"No hidden test — dry-run on {n} train files\")\n    test_paths = sorted((BASE / \"train_soundscapes\").glob(\"*.ogg\"))[:n]\nelse:\n    print(f\"Hidden test files: {len(test_paths)}\")\nmeta_te, sc_te, emb_te = run_perch(test_paths, CFG[\"batch_files\"], verbose=CFG[\"verbose\"])\nprint(f\"Test scores: {sc_te.shape}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sigmoid(x):\n    return 1.0 / (1.0 + np.exp(-np.clip(x, -30, 30)))\n\n# Train ProtoSSM — UPGRADED d_model=256\nt0 = time.time()\nproto_model, site2i_tr = train_light_proto_ssm(\n    emb_tr, sc_tr, Y_FULL_aligned, meta_tr,\n    n_epochs=CFG[\"proto_ssm_train\"][\"n_epochs\"],\n    patience=CFG[\"proto_ssm_train\"][\"patience\"],\n    lr=CFG[\"proto_ssm_train\"][\"lr\"],\n    d_model=CFG[\"proto_ssm_train\"][\"d_model\"],   # 256\n    verbose=False)\nprint(f\"ProtoSSM training: {time.time()-t0:.1f}s\")\n\nn_test_files  = len(sc_te) // N_WINDOWS\nemb_te_f      = emb_te.reshape(n_test_files, N_WINDOWS, -1)\nsc_te_f       = sc_te.reshape(n_test_files, N_WINDOWS, -1)\ntest_fnames   = meta_te.drop_duplicates(\"filename\")[\"filename\"].tolist()\nn_sites_cap   = 20\ntest_site_ids = np.array([\n    min(site2i_tr.get(meta_te.loc[meta_te[\"filename\"]==fn,\"site\"].iloc[0], 0), n_sites_cap-1)\n    for fn in test_fnames], dtype=np.int64)\ntest_hour_ids = np.array([\n    int(meta_te.loc[meta_te[\"filename\"]==fn,\"hour_utc\"].iloc[0]) % 24\n    for fn in test_fnames], dtype=np.int64)\n\nproto_model.eval()\nwith torch.no_grad():\n    proto_out = proto_model(\n        torch.tensor(emb_te_f, dtype=torch.float32),\n        torch.tensor(sc_te_f,  dtype=torch.float32),\n        site_ids=torch.tensor(test_site_ids, dtype=torch.long),\n        hours   =torch.tensor(test_hour_ids, dtype=torch.long),\n    ).numpy()\nproto_scores_flat = proto_out.reshape(-1, N_CLASSES).astype(np.float32)\nprint(f\"ProtoSSM test done: {proto_scores_flat.shape}\")\n\nprior_tables   = build_prior_tables(sc, Y_SC)\nsc_te_adjusted = apply_prior(sc_te, sites=meta_te[\"site\"].to_numpy(),\n                              hours=meta_te[\"hour_utc\"].to_numpy(), tables=prior_tables, lambda_prior=0.4)\n\nprobe_models, emb_scaler, emb_pca, alpha_blend = train_mlp_probes(\n    emb=emb_tr, scores_raw=sc_tr, Y=Y_FULL_aligned,\n    min_pos=5, pca_dim=CFG[\"mlp_pca_dim\"], alpha_blend=0.4)  # pca=128\n\nsc_te_adjusted = apply_mlp_probes_vectorized(\n    emb_te, sc_te_adjusted, probe_models, emb_scaler, emb_pca, alpha_blend)\n\nENSEMBLE_W      = 0.5\nfirst_pass_flat = (ENSEMBLE_W * proto_scores_flat + (1.0 - ENSEMBLE_W) * sc_te_adjusted)\n\nn_tr_files    = len(sc_tr) // N_WINDOWS\nemb_tr_f      = emb_tr.reshape(n_tr_files, N_WINDOWS, -1)\nsc_tr_f       = sc_tr.reshape(n_tr_files, N_WINDOWS, -1)\ntr_fnames     = meta_tr.drop_duplicates(\"filename\")[\"filename\"].tolist()\ntr_site_ids   = np.array([\n    min(site2i_tr.get(meta_tr.loc[meta_tr[\"filename\"]==fn,\"site\"].iloc[0], 0), n_sites_cap-1)\n    for fn in tr_fnames], dtype=np.int64)\ntr_hour_ids   = np.array([\n    int(meta_tr.loc[meta_tr[\"filename\"]==fn,\"hour_utc\"].iloc[0]) % 24\n    for fn in tr_fnames], dtype=np.int64)\n\nproto_tr_out = run_tta_proto(\n    proto_model, emb_tr_f, sc_tr_f,\n    site_t=torch.tensor(tr_site_ids, dtype=torch.long),\n    hour_t=torch.tensor(tr_hour_ids, dtype=torch.long),\n    shifts=CFG[\"tta_shifts\"],\n)\nproto_tr_flat = proto_tr_out.reshape(-1, N_CLASSES).astype(np.float32)\nsc_tr_prior   = apply_prior(sc_tr, sites=meta_tr[\"site\"].to_numpy(),\n                             hours=meta_tr[\"hour_utc\"].to_numpy(), tables=prior_tables, lambda_prior=0.4)\nsc_tr_mlp = apply_mlp_probes_vectorized(emb_tr, sc_tr_prior, probe_models, emb_scaler, emb_pca, alpha_blend)\nfirst_pass_tr = (ENSEMBLE_W * proto_tr_flat + (1.0 - ENSEMBLE_W) * sc_tr_mlp)\n\ntrain_probs_for_calib = sigmoid(first_pass_tr)\nPER_CLASS_THRESHOLDS = calibrate_and_optimize_thresholds(\n    oof_probs=train_probs_for_calib, Y_FULL=Y_FULL_aligned,\n    threshold_grid=[0.25, 0.30, 0.35, 0.40, 0.45, 0.50, 0.55, 0.60, 0.65, 0.70], n_windows=N_WINDOWS)\n\nt0 = time.time()\nres_model, correction_weight = train_residual_ssm(\n    emb_full=emb_tr, first_pass_flat=first_pass_tr, Y_full=Y_FULL_aligned,\n    site_ids=tr_site_ids, hour_ids=tr_hour_ids,\n    n_epochs=30, patience=8, lr=1e-3, correction_weight=0.30, verbose=False)\nprint(f\"ResidualSSM training: {time.time()-t0:.1f}s\")\n\nfirst_pass_te_f  = first_pass_flat.reshape(n_test_files, N_WINDOWS, -1)\nres_model.eval()\nwith torch.no_grad():\n    test_correction = res_model(\n        torch.tensor(emb_te_f,        dtype=torch.float32),\n        torch.tensor(first_pass_te_f, dtype=torch.float32),\n        site_ids=torch.tensor(test_site_ids, dtype=torch.long),\n        hours   =torch.tensor(test_hour_ids, dtype=torch.long),\n    ).numpy()\ncorrection_flat = test_correction.reshape(-1, N_CLASSES).astype(np.float32)\nfinal_scores    = (first_pass_flat + correction_weight * correction_flat)\nfinal_scores = final_scores / temperatures[None, :]\nprobs = sigmoid(final_scores)\nprobs = file_confidence_scale(probs, n_windows=N_WINDOWS, top_k=2, power=0.4)\nprobs = rank_aware_scaling(   probs, n_windows=N_WINDOWS, power=0.4)\nprobs = adaptive_delta_smooth(probs, n_windows=N_WINDOWS, base_alpha=0.20)\nprobs = np.clip(probs, 0.0, 1.0)\nprobs = apply_per_class_thresholds(probs, PER_CLASS_THRESHOLDS)\nsub = pd.DataFrame(probs.astype(np.float32), columns=PRIMARY_LABELS)\nsub.insert(0, \"row_id\", meta_te[\"row_id\"].values)\nsub.to_csv(\"submission_protossm.csv\", index=False)\nprint(\"ProtoSSM execution complete\")\nprint(f\"Wall: {(time.time()-_WALL_START)/60:.1f} min\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import librosa\nfrom scipy.ndimage import gaussian_filter1d\nN_MELS_SED = 256; N_FFT_SED = 2048; HOP_SED = 512; FMIN_SED = 20; FMAX_SED = 16000; TOP_DB_SED = 80\ndef find_sed_dir():\n    hits = sorted(Path(\"/kaggle/input\").rglob(\"sed_fold0.onnx\"))\n    if not hits: raise FileNotFoundError(\"sed_fold0.onnx not found.\")\n    return hits[0].parent\ndef make_sed_session(path):\n    so = ort.SessionOptions(); so.intra_op_num_threads = 4; so.inter_op_num_threads = 1\n    so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL\n    return ort.InferenceSession(str(path), sess_options=so, providers=[\"CPUExecutionProvider\"])\ndef audio_to_mel(chunks):\n    mels = []\n    for x in chunks:\n        s = librosa.feature.melspectrogram(y=x, sr=SR, n_fft=N_FFT_SED, hop_length=HOP_SED,\n                                           n_mels=N_MELS_SED, fmin=FMIN_SED, fmax=FMAX_SED, power=2.0)\n        s = librosa.power_to_db(s, top_db=TOP_DB_SED); s = (s - s.mean()) / (s.std() + 1e-6); mels.append(s)\n    return np.stack(mels)[:, None].astype(np.float32)\ndef file_to_sed_chunks(path):\n    y, sr0 = sf.read(str(path), dtype=\"float32\", always_2d=False)\n    if y.ndim == 2: y = y.mean(axis=1)\n    if sr0 != SR: y = librosa.resample(y, orig_sr=sr0, target_sr=SR)\n    n = 60 * SR\n    if len(y) < n: y = np.pad(y, (0, n - len(y)))\n    else: y = y[:n]\n    chunks = y.reshape(N_WINDOWS, WINDOW_SAMPLES); ends = np.arange(1, N_WINDOWS + 1) * WINDOW_SEC\n    return chunks, ends\ndef sigmoid_sed(x): return (1.0 / (1.0 + np.exp(-np.clip(x, -50, 50)))).astype(np.float32)\nsed_dir = find_sed_dir()\nsed_fold_paths = sorted(sed_dir.glob(\"sed_fold*.onnx\"), key=lambda p: int(re.search(r\"sed_fold(\\d+)\", p.name).group(1)))\nsed_sessions = [make_sed_session(p) for p in sed_fold_paths]\nprint(f\"SED folds: {[p.name for p in sed_fold_paths]}\")\nsed_rows, sed_preds = [], []\nfor i, path in enumerate(test_paths, 1):\n    chunks, ends = file_to_sed_chunks(path)\n    mel = audio_to_mel(chunks)\n    p_sum = np.zeros((len(chunks), N_CLASSES), dtype=np.float32)\n    for sess in sed_sessions:\n        outs = sess.run(None, {sess.get_inputs()[0].name: mel})\n        clip_logits = outs[0]; frame_max = outs[1].max(axis=1)\n        p_sum += 0.5 * sigmoid_sed(clip_logits) + 0.5 * sigmoid_sed(frame_max)\n    p_mean = p_sum / len(sed_sessions)\n    if len(p_mean) > 1:\n        p_mean = gaussian_filter1d(p_mean, sigma=0.65, axis=0, mode=\"nearest\").astype(np.float32)\n    stem = path.stem; sed_rows.extend([f\"{stem}_{int(t)}\" for t in ends]); sed_preds.append(p_mean)\n    if i == 1 or i % 50 == 0 or i == len(test_paths): print(f\"SED: {i}/{len(test_paths)}\")\nsed_preds_arr = np.concatenate(sed_preds, axis=0)\nsed_sub = pd.DataFrame(np.clip(sed_preds_arr, 0.0, 1.0), columns=PRIMARY_LABELS)\nsed_sub.insert(0, \"row_id\", sed_rows)\nsed_sub.to_csv(\"submission_sed.csv\", index=False)\nprint(\"Distilled SED Processing Complete.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Student CNN (EfficientNet-B4 + GeM, 5 folds, best AUC=0.8988) ──\nimport torchaudio.transforms as _TS\nimport torch.nn.functional as _FS\n\n_STU_DIR = Path(\"/kaggle/input/datasets/eslamelokpy/birdclef2026-student-onnx\")\n_STU_ONNX = sorted(_STU_DIR.glob(\"student_fold*.onnx\"))\nprint(f\"Student folds: {len(_STU_ONNX)}\")\n\n_student_preds = None\nif len(_STU_ONNX) > 0:\n    # Fold weights by validation AUC (fold4 best=0.8988)\n    _FOLD_AUCS = {1:0.8513, 2:0.8482, 3:0.8337, 4:0.8988, 5:0.8649}\n    _stu_sessions, _stu_weights = [], []\n    for _fp in _STU_ONNX:\n        _fold_n = int(_fp.stem.split(\"fold\")[-1])\n        _so = ort.SessionOptions()\n        _so.intra_op_num_threads = 4\n        _so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL\n        _stu_sessions.append(ort.InferenceSession(str(_fp), sess_options=_so,\n                             providers=[\"CPUExecutionProvider\"]))\n        _stu_weights.append(_FOLD_AUCS.get(_fold_n, 0.85))\n    _stu_w = np.array(_stu_weights); _stu_w = _stu_w / _stu_w.sum()\n    print(f\"Fold weights: {dict(zip([1,2,3,4,5], _stu_w.round(3)))}\")\n\n    # EXACT same mel as training: img=256, n_mels=256, n_fft=2048, hop=512\n    _IMG_STU = 256\n    _stu_mel = _TS.MelSpectrogram(sample_rate=SR, n_fft=2048, hop_length=512,\n                                    n_mels=256, f_min=20, f_max=16000, power=2.0)\n    _stu_db  = _TS.AmplitudeToDB(top_db=80)\n\n    def _stu_chunk_to_mel(wav_np):\n        t = torch.from_numpy(wav_np).float()\n        m = _stu_db(_stu_mel(t))\n        mn, sd = m.mean(), m.std()\n        m = (m - mn) / (sd + 1e-6)\n        m = _FS.interpolate(m.unsqueeze(0).unsqueeze(0),\n                             size=(_IMG_STU, _IMG_STU),\n                             mode=\"bilinear\", align_corners=False).squeeze()\n        return m.repeat(3, 1, 1).unsqueeze(0).numpy().astype(\"float32\")\n\n    _N_TE = len(test_paths)\n    _student_preds = np.zeros((_N_TE * N_WINDOWS, N_CLASSES), np.float32)\n    _inp_stu = _stu_sessions[0].get_inputs()[0].name\n    _t0_stu  = time.time()\n\n    for _fi, _path in enumerate(tqdm(test_paths, \"Student CNN\")):\n        try:\n            _y, _ = sf.read(str(_path), dtype=\"float32\", always_2d=False)\n            if _y.ndim == 2: _y = _y.mean(1)\n            if len(_y) < FILE_SAMPLES: _y = np.pad(_y, (0, FILE_SAMPLES-len(_y)))\n            else: _y = _y[:FILE_SAMPLES]\n        except: _y = np.zeros(FILE_SAMPLES, np.float32)\n\n        _chunks = _y.reshape(N_WINDOWS, WINDOW_SAMPLES)\n        _batch  = np.concatenate([_stu_chunk_to_mel(_c) for _c in _chunks], axis=0)\n        _p_sum  = np.zeros((N_WINDOWS, N_CLASSES), np.float32)\n\n        for _si, (_sess, _w) in enumerate(zip(_stu_sessions, _stu_w)):\n            _logits = _sess.run(None, {_inp_stu: _batch})[0]\n            _p_sum += _w * (1.0 / (1.0 + np.exp(-_logits.astype(np.float32))))\n\n        _student_preds[_fi*N_WINDOWS:(_fi+1)*N_WINDOWS] = _p_sum\n\n    _elapsed = time.time() - _t0_stu\n    print(f\"Student done: {_elapsed/60:.1f}min | range [{_student_preds.min():.3f},{_student_preds.max():.3f}]\")\n    _stu_df = pd.DataFrame(np.clip(_student_preds, 0, 1).astype(np.float32), columns=PRIMARY_LABELS)\n    _stu_df.insert(0, \"row_id\", meta_te[\"row_id\"].values)\n    _stu_df.to_csv(\"submission_student.csv\", index=False)\n    print(\"submission_student.csv saved\")\nelse:\n    print(\"Student ONNX not found — 2-way blend only\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_hgn_preds = None","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, numpy as np, pandas as pd\nfrom pathlib import Path\nEPS = 1e-5\ndf_proto = pd.read_csv(\"submission_protossm.csv\")\ndf_sed   = pd.read_csv(\"submission_sed.csv\")\ncols = [c for c in df_proto.columns if c != \"row_id\"]\ndf_sed = df_sed.set_index(\"row_id\").loc[df_proto[\"row_id\"]].reset_index()\np_proto = np.clip(df_proto[cols].to_numpy(np.float32), EPS, 1-EPS)\np_sed   = np.clip(df_sed  [cols].to_numpy(np.float32), EPS, 1-EPS)\nrank_proto = pd.DataFrame(p_proto).rank(axis=0, pct=True).to_numpy(np.float32)\nrank_sed   = pd.DataFrame(p_sed  ).rank(axis=0, pct=True).to_numpy(np.float32)\n\ntry:\n    df_stu = pd.read_csv(\"submission_student.csv\")\n    df_stu = df_stu.set_index(\"row_id\").loc[df_proto[\"row_id\"]].reset_index()\n    p_stu    = np.clip(df_stu[cols].to_numpy(np.float32), EPS, 1-EPS)\n    rank_stu = pd.DataFrame(p_stu).rank(axis=0, pct=True).to_numpy(np.float32)\n    pred = 0.50 * rank_proto + 0.30 * rank_sed + 0.20 * rank_stu\n    print(\"3-way: 50% ProtoSSM + 30% SED + 20% Student (B4+GeM, fold4 AUC=0.8988)\")\nexcept Exception as _e:\n    pred = 0.60 * rank_proto + 0.40 * rank_sed\n    print(f\"2-way fallback: {_e}\")\n\nrow_ids  = df_proto[\"row_id\"].astype(str).to_numpy()\nfile_ids = np.array([\"_\".join(r.split(\"_\")[:-1]) for r in row_ids])\n\n# Rescue logic (proven 0.946)\nfake_only = (p_proto > 0.50) & (p_sed < 0.05)\npred = np.where(fake_only, (1.0-0.08)*pred + 0.08*rank_proto, pred)\n\noffs = np.arange(-3,4,dtype=np.float32)\npk   = (1.0+(offs/1.20)**2/2.0)**(-1.5); pk /= pk.sum()\npa_ctx = p_proto.copy()\nfor fid in pd.unique(file_ids):\n    m = file_ids==fid; x = p_proto[m]\n    if len(x)>1:\n        xp = np.pad(x,((3,3),(0,0)),mode=\"edge\")\n        pa_ctx[m] = sum(pk[i]*xp[i:i+len(x)] for i in range(7))\nxctx = pd.DataFrame(pa_ctx).rank(axis=0,pct=True).to_numpy(np.float32)\nproto_cont=(xctx>0.88)&(rank_proto>0.75)&(p_sed<0.12)&(~fake_only)\npred = np.where(proto_cont,(1.0-0.15)*pred+0.15*np.maximum(rank_proto,xctx),pred)\n\nsed_only=(rank_sed>0.95)&(rank_proto<0.80)&(~fake_only)&(~proto_cont)\npred = np.where(sed_only,(1.0-0.12)*pred+0.12*rank_sed,pred)\n\nsub = df_proto.copy(); sub[cols] = pred.astype(np.float32)\n\nMIRROR_PAIRS=((\"47158son15\",\"47158son16\"),(\"47158son09\",\"47158son12\"),\n              (\"47158son02\",\"47158son14\"),(\"47158son13\",\"47158son21\",\"47158son22\",\"47158son23\"))\ncol_to_idx={l:i for i,l in enumerate(cols)}\nfor group in MIRROR_PAIRS:\n    vi=[col_to_idx[s] for s in group if s in col_to_idx]\n    if len(vi)>=2:\n        gmax=sub[cols].iloc[:,vi].max(axis=1).to_numpy(np.float32)\n        for idx in vi: sub.iloc[:,idx+1]=gmax\n\ntry:\n    tax_df=pd.read_csv(BASE/\"taxonomy.csv\").set_index(\"primary_label\")\n    rare={\"Amphibia\",\"Mammalia\",\"Reptilia\"}\n    for ci,sp in enumerate(cols):\n        if sp in tax_df.index and tax_df.loc[sp,\"class_name\"] in rare:\n            vals=sub.iloc[:,ci+1].to_numpy(np.float32)\n            sub.iloc[:,ci+1]=np.where(vals<vals.mean()+0.05,vals*0.9,vals)\nexcept: pass\n\nsub.to_csv(\"submission.csv\",index=False)\nprint(f\"submission.csv: {sub.shape}\")\nprint(f\"Total wall: {(time.time()-_WALL_START)/60:.1f} min\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}