{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"datasetVersion","sourceId":8209908,"datasetId":4865209,"databundleVersionId":8334538},{"sourceType":"modelInstanceVersion","sourceId":32637,"databundleVersionId":8261530,"modelInstanceId":26649,"modelId":37756},{"sourceType":"modelInstanceVersion","sourceId":516989,"databundleVersionId":13353982,"modelInstanceId":404337,"modelId":319}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Day 1: Setup and Smoketest\nThis notebook covers Kaggle dataset setup, eval-driven species curation, and a smoke test of BirdNET, Perch, and AVES.","metadata":{}},{"cell_type":"code","source":"!pip install -q birdnetlib transformers audiomentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T15:09:24.193190Z","iopub.execute_input":"2026-04-25T15:09:24.193452Z","iopub.status.idle":"2026-04-25T15:09:32.166260Z","shell.execute_reply.started":"2026-04-25T15:09:24.193431Z","shell.execute_reply":"2026-04-25T15:09:32.165455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\n\n# Setup paths dynamically by searching /kaggle/input\nINPUT_DIR = Path('/kaggle/input')\nDATA_OUT = Path('./data')\nDATA_OUT.mkdir(exist_ok=True)\n\n# Find the files automatically\ntry:\n    TRAIN_CSV = next(INPUT_DIR.rglob('train_metadata.csv'))\n    print(f\"Found Train CSV: {TRAIN_CSV}\")\nexcept StopIteration:\n    TRAIN_CSV = Path('not_found')\n\neval_labels_paths = [p for p in INPUT_DIR.rglob('*.csv') if 'train_metadata' not in p.name and 'test' not in p.name]\nEVAL_LABELS_CSV = eval_labels_paths[0] if eval_labels_paths else Path('not_found')\nif EVAL_LABELS_CSV.exists():\n    print(f\"Found Eval CSV: {EVAL_LABELS_CSV}\")\n\n# 1. Eval-driven curation\nfocus_from_eval = []\nif EVAL_LABELS_CSV.exists():\n    eval_df = pd.read_csv(EVAL_LABELS_CSV)\n    print(f'Eval dataset shape: {eval_df.shape}')\n    \n    # Get species columns (excluding row_id, audio_id, etc)\n    species_cols = [c for c in eval_df.columns if c not in ['row_id', 'audio_id', 'time', 'site']]\n    eval_counts = eval_df[species_cols].sum().sort_values(ascending=False)\n    focus_from_eval = eval_counts[eval_counts > 0].index.tolist()\n    print(f'Found {len(focus_from_eval)} species with >=1 positive in eval.')\n    \n    eval_counts.to_frame('eval_positives').to_csv(DATA_OUT / 'eval_coverage.csv')\nelse:\n    print('Eval dataset not found.')\n\n# 2. Focus Species Generation\nif TRAIN_CSV.exists():\n    train_df = pd.read_csv(TRAIN_CSV)\n    print(f'Train metadata shape: {train_df.shape}')\n    \n    train_counts = train_df['primary_label'].value_counts()\n    focus_set = set(focus_from_eval)\n    \n    # Pad up to 50 using the most abundant training species\n    target_len = 50\n    for sp in train_counts.index:\n        if len(focus_set) >= target_len:\n            break\n        focus_set.add(sp)\n        \n    focus_list = sorted(list(focus_set))\n    pd.DataFrame({'species': focus_list}).to_csv(DATA_OUT / 'species_focus.csv', index=False)\n    print(f'Selected {len(focus_list)} focus species and saved to data/species_focus.csv')\nelse:\n    print('Train metadata not found.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T15:16:32.122541Z","iopub.execute_input":"2026-04-25T15:16:32.123322Z","iopub.status.idle":"2026-04-25T15:18:20.570364Z","shell.execute_reply.started":"2026-04-25T15:16:32.123287Z","shell.execute_reply":"2026-04-25T15:18:20.569388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile extract_embeddings.py\nimport librosa\nimport numpy as np\nimport tensorflow as tf\nfrom huggingface_hub import snapshot_download\nfrom birdnetlib import Recording\nfrom birdnetlib.analyzer import Analyzer\nfrom transformers import AutoModel, AutoFeatureExtractor\nimport torch\n\nSR = 32000\nCHUNK_SEC = 5.0\n\ndef load_audio(path: str, sr: int = SR) -> np.ndarray:\n    try:\n        y, _ = librosa.load(path, sr=sr, mono=True)\n        return y\n    except Exception as e:\n        print(f\"Error loading {path}: {e}\")\n        return np.array([])\n\n# BirdNET\ntry:\n    analyzer = Analyzer()\nexcept Exception as e:\n    analyzer = None\n\ndef birdnet_predict(audio_path: str, lat=10.5, lon=76.5, week=-1):\n    if not analyzer: return []\n    rec = Recording(analyzer, audio_path, lat=lat, lon=lon, week=week, min_conf=0.0, overlap=0.0)\n    rec.analyze()\n    return rec.detections\n\n# Perch\ntry:\n    PERCH_REPO = \"cgeorgiaw/Perch\"\n    local_perch = snapshot_download(repo_id=PERCH_REPO)\n    perch = tf.saved_model.load(local_perch)\nexcept Exception as e:\n    perch = None\n\ndef perch_embed(audio_5s_32k: np.ndarray) -> np.ndarray:\n    if not perch: return np.zeros(1536)\n    x = tf.constant(audio_5s_32k[None, :], dtype=tf.float32)\n    out = perch.infer_tf(x)\n    return out[\"embedding\"].numpy()[0]\n\n# AVES\ntry:\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    aves = AutoModel.from_pretrained(\"earthspecies/aves-base-bio\").eval().to(device)\n    aves_fx = AutoFeatureExtractor.from_pretrained(\"earthspecies/aves-base-bio\")\nexcept Exception as e:\n    aves = None\n\n@torch.no_grad()\ndef aves_embed(audio_5s_16k: np.ndarray) -> np.ndarray:\n    if not aves: return np.zeros(768)\n    x = aves_fx(audio_5s_16k, sampling_rate=16000, return_tensors=\"pt\").input_values.to(device)\n    h = aves(x).last_hidden_state.mean(dim=1)\n    return h.cpu().numpy()[0]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T15:20:15.431611Z","iopub.execute_input":"2026-04-25T15:20:15.432310Z","iopub.status.idle":"2026-04-25T15:20:15.438271Z","shell.execute_reply.started":"2026-04-25T15:20:15.432280Z","shell.execute_reply":"2026-04-25T15:20:15.437444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import extract_embeddings\nprint(\"Models loaded successfully! Smoke test passed.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T15:20:27.891162Z","iopub.execute_input":"2026-04-25T15:20:27.891853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}