{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"datasetVersion","sourceId":8209908,"datasetId":4865209,"databundleVersionId":8334538},{"sourceType":"modelInstanceVersion","sourceId":32637,"databundleVersionId":8261530,"modelInstanceId":26649,"modelId":37756},{"sourceType":"modelInstanceVersion","sourceId":516989,"databundleVersionId":13353982,"modelInstanceId":404337,"modelId":319},{"sourceType":"kernelVersion","sourceId":314394777},{"sourceType":"kernelVersion","sourceId":314404207}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Day 3: Soundscape Benchmark\nThis notebook focuses on processing the 100 labeled real-world soundscapes and computing the domain shift performance drop.","metadata":{}},{"cell_type":"code","source":"!pip install -q --upgrade \"tensorflow[and-cuda]>=2.16.1\" birdnetlib pyarrow scikit-learn resampy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile extract_embeddings.py\nimport librosa\nimport numpy as np\nimport tensorflow as tf\nfrom birdnetlib import Recording\nfrom birdnetlib.analyzer import Analyzer\n\nSR = 32000\nCHUNK_SEC = 5.0\n\ndef load_audio(path: str, sr: int = SR) -> np.ndarray:\n    try:\n        y, _ = librosa.load(path, sr=sr, mono=True)\n        return y\n    except Exception as e:\n        print(f\"Error loading {path}: {e}\")\n        return np.array([])\n\ndef chunks(y: np.ndarray, sec: float = CHUNK_SEC, sr: int = SR, overlap: float = 0.0):\n    step = int(sec * sr * (1 - overlap))\n    win = int(sec * sr)\n    for i in range(0, max(1, len(y) - win + 1), step):\n        c = y[i : i + win]\n        if len(c) < win:\n            c = np.pad(c, (0, win - len(c)))\n        yield c\n\n# -----------------\n# Offline BirdNET\n# -----------------\ntry:\n    analyzer = Analyzer()\nexcept Exception as e:\n    analyzer = None\n\ndef birdnet_predict(audio_path: str, lat=10.5, lon=76.5):\n    if not analyzer: return []\n    rec = Recording(analyzer, audio_path, lat=lat, lon=lon, min_conf=0.0)\n    try:\n        rec.analyze()\n        return rec.detections\n    except:\n        return []\n\n# -----------------\n# Offline Perch v2\n# -----------------\ntry:\n    from pathlib import Path\n    pb_path = next(Path(\"/kaggle/input\").rglob(\"saved_model.pb\"))\n    PERCH_MODEL_PATH = str(pb_path.parent)\n    _raw_model = tf.saved_model.load(PERCH_MODEL_PATH)\n    perch = _raw_model.signatures[\"serving_default\"]\nexcept Exception as e:\n    print(f\"Perch init error: {e}. Did you remember to attach the Kaggle Model?\")\n    perch = None\n\ndef perch_embed(audio_5s_32k: np.ndarray) -> np.ndarray:\n    if not perch: return np.zeros(1536)\n    x = tf.constant(audio_5s_32k[None, :], dtype=tf.float32)\n    out = perch(inputs=x)\n    return out[\"embedding\"].numpy()[0]\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport ast\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport extract_embeddings\nfrom tqdm.auto import tqdm\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\n\nINPUT_DIR = Path('/kaggle/input')\nDATA_OUT = Path('./data')\nDATA_OUT.mkdir(exist_ok=True)\nCACHE_OUT = Path('./cache/embeddings')\nCACHE_OUT.mkdir(parents=True, exist_ok=True)\nREPORTS_OUT = Path('./reports')\nREPORTS_OUT.mkdir(exist_ok=True)\n\ntry:\n    EVAL_LABELS_CSV = next(INPUT_DIR.rglob('labeled_soundscapes.csv'))\n    eval_df = pd.read_csv(EVAL_LABELS_CSV)\n    print(f\"Loaded soundscape labels: {eval_df.shape}\")\nexcept StopIteration:\n    print(\"Could not find labeled_soundscapes.csv! Make sure richolson dataset is attached.\")\n    eval_df = pd.DataFrame()\n\ntry:\n    focus_df = pd.read_csv(next(INPUT_DIR.rglob('species_focus.csv')))\n    focus_species = focus_df['species'].tolist()\n    print(f\"Loaded {len(focus_species)} focus species.\")\nexcept StopIteration:\n    print(\"Could not find species_focus.csv!\")\n    focus_species = []\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter eval labels to focus species\nif not eval_df.empty and len(focus_species) > 0:\n    # Check if audio_id and time exist. If not, extract them from row_id\n    if 'audio_id' not in eval_df.columns and 'row_id' in eval_df.columns:\n        # row_id format is usually {audio_id}_{time_end}\n        eval_df['audio_id'] = eval_df['row_id'].apply(lambda x: x.rsplit('_', 1)[0])\n        eval_df['time'] = eval_df['row_id'].apply(lambda x: int(x.rsplit('_', 1)[1]))\n        \n    # Get the species columns\n    species_cols = [c for c in eval_df.columns if c not in ['row_id', 'audio_id', 'time', 'site']]\n    available_focus = [c for c in focus_species if c in species_cols]\n    \n    # Melt eval_df to (audio_id, time, species_code, label)\n    eval_melt = eval_df.melt(id_vars=['audio_id', 'time'], value_vars=available_focus, var_name='species_code', value_name='label')\n    print(f\"Eval dataset melted for focus species. Shape: {eval_melt.shape}\")\n\n    \n        # ... (top of the cell remains the same) ...\n    print(f\"Eval dataset melted for focus species. Shape: {eval_melt.shape}\")\n    \n    unique_audios = eval_df['audio_id'].unique()\n    print(f\"Found {len(unique_audios)} unique soundscape files to process.\")\n    \n    # Find exactly where the 100 labeled soundscape audio files are located\n    first_audio_filename = f\"{unique_audios[0]}.ogg\"\n    \n    try:\n        sample_audio_path = next(INPUT_DIR.rglob(first_audio_filename))\n        unlabeled_audio_dir = sample_audio_path.parent\n        print(f\"Found correct audio directory: {unlabeled_audio_dir}\")\n    except StopIteration:\n        print(f\"Could not find {first_audio_filename}! Make sure the dataset is attached.\")\n        unlabeled_audio_dir = None\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 'unlabeled_audio_dir' in locals() and unlabeled_audio_dir:\n    results_perch_ss = []\n    results_birdnet_ss = []\n    \n    for audio_id in tqdm(unique_audios):\n        audio_filename = f\"{audio_id}.ogg\"\n        file_path = unlabeled_audio_dir / audio_filename\n        if not file_path.exists():\n            continue\n            \n        y = extract_embeddings.load_audio(str(file_path))\n        if len(y) == 0: continue\n            \n        # Get all 5-second chunks\n        chunk_gen = list(extract_embeddings.chunks(y))\n        \n        # Run BirdNET once for the whole file\n        detections = extract_embeddings.birdnet_predict(str(file_path))\n        results_birdnet_ss.append({\n            'audio_id': audio_id,\n            'detections': str(detections)\n        })\n        \n        # Run Perch on each chunk\n        for i, chunk in enumerate(chunk_gen):\n            time_end = (i + 1) * 5\n            emb = extract_embeddings.perch_embed(chunk)\n            results_perch_ss.append({\n                'audio_id': audio_id,\n                'time': time_end,\n                'emb': emb\n            })\n            \n    # Save soundscape embeddings\n    pd.DataFrame(results_perch_ss).to_parquet(CACHE_OUT / 'perch_soundscape.parquet')\n    pd.DataFrame(results_birdnet_ss).to_parquet(CACHE_OUT / 'birdnet_soundscape.parquet')\n    print(\"Soundscape extraction complete and saved!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate Domain Shift (Soundscapes)\nif 'results_perch_ss' in locals() and len(results_perch_ss) > 0:\n    print(\"\\n--- Evaluating Soundscape Benchmark ---\")\n    \n    perch_ss_df = pd.DataFrame(results_perch_ss)\n    birdnet_ss_df = pd.DataFrame(results_birdnet_ss)\n    \n    # Load Day 2 Train parquets to build prototypes\n    train_perch_files = list(INPUT_DIR.rglob('perch_train_*.parquet'))\n    if train_perch_files:\n        train_perch_df = pd.concat([pd.read_parquet(f) for f in train_perch_files]).reset_index(drop=True)\n        \n        prototypes = {}\n        for sp in available_focus:\n            sp_embs = np.stack(train_perch_df[train_perch_df['species_code'] == sp]['emb'].values)\n            if len(sp_embs) > 0:\n                prototypes[sp] = np.mean(sp_embs, axis=0)\n                \n        valid_ss_species = list(prototypes.keys())\n        P_matrix = np.stack([prototypes[sp] for sp in valid_ss_species])\n        \n        # Perch Prototype Scoring\n        perch_ss_preds = []\n        for idx, row in perch_ss_df.iterrows():\n            emb = row['emb']\n            sims = np.dot(P_matrix, emb) / (np.linalg.norm(P_matrix, axis=1) * np.linalg.norm(emb) + 1e-12)\n            sims_scaled = sims / 0.05\n            probs = np.exp(sims_scaled - np.max(sims_scaled)) / np.sum(np.exp(sims_scaled - np.max(sims_scaled)))\n            perch_ss_preds.append(probs)\n            \n        perch_ss_preds = np.array(perch_ss_preds)\n        \n        # Format eval labels properly\n        y_true_ss = []\n        y_pred_perch_ss = []\n        \n        for idx, row in perch_ss_df.iterrows():\n            audio = row['audio_id']\n            time = row['time']\n            labels_for_chunk = eval_df[(eval_df['audio_id'] == audio) & (eval_df['time'] == time)]\n            \n            if len(labels_for_chunk) > 0:\n                y_t = np.zeros(len(valid_ss_species))\n                for i, sp in enumerate(valid_ss_species):\n                    if sp in labels_for_chunk.columns:\n                        y_t[i] = labels_for_chunk.iloc[0][sp]\n                \n                y_true_ss.append(y_t)\n                y_pred_perch_ss.append(perch_ss_preds[idx])\n                \n        y_true_ss = np.array(y_true_ss)\n        y_pred_perch_ss = np.array(y_pred_perch_ss)\n        \n        perch_auc_dict = {}\n        for i, sp in enumerate(valid_ss_species):\n            if np.sum(y_true_ss[:, i]) > 0 and len(np.unique(y_true_ss[:, i])) > 1:\n                perch_auc_dict[sp] = roc_auc_score(y_true_ss[:, i], y_pred_perch_ss[:, i])\n                \n        perch_macro_auc = np.mean(list(perch_auc_dict.values())) if perch_auc_dict else np.nan\n        print(f\"Perch Prototype Soundscape Macro AUC: {perch_macro_auc:.4f}\")\n        \n        # BirdNET Zero-Shot Scoring\n        taxa_csv = next(INPUT_DIR.rglob('eBird_Taxonomy_v2021.csv'), None)\n        taxa_mapping = {}\n        if taxa_csv:\n            taxa_df = pd.read_csv(taxa_csv)\n            taxa_mapping = dict(zip(taxa_df['SCI_NAME'].str.lower(), taxa_df['SPECIES_CODE']))\n            \n        y_pred_birdnet_ss = np.zeros_like(y_true_ss)\n        \n        for idx, row in birdnet_ss_df.iterrows():\n            audio = row['audio_id']\n            try:\n                dets = ast.literal_eval(row['detections'])\n                for d in dets:\n                    sci_name = d.get('scientific_name', '').lower()\n                    conf = d.get('confidence', 0.0)\n                    start_t = d.get('start_time', 0.0)\n                    end_t = d.get('end_time', 5.0)\n                    \n                    # Find matching chunk index based on time\n                    time_end = int(np.ceil(end_t / 5.0) * 5)\n                    \n                    sp_code = taxa_mapping.get(sci_name, None)\n                    if sp_code in valid_ss_species:\n                        s_idx = valid_ss_species.index(sp_code)\n                        \n                        # Locate the row in y_pred_birdnet_ss corresponding to this audio & time\n                        # We need to map back to the flat true list.\n                        # This is an approximation for matching indices\n                        match_indices = [i for i, (a, t) in enumerate(zip(perch_ss_df['audio_id'], perch_ss_df['time'])) if a == audio and t == time_end]\n                        for m_idx in match_indices:\n                            if m_idx < len(y_pred_birdnet_ss):\n                                y_pred_birdnet_ss[m_idx, s_idx] = max(y_pred_birdnet_ss[m_idx, s_idx], conf)\n            except:\n                pass\n                \n        birdnet_auc_dict = {}\n        for i, sp in enumerate(valid_ss_species):\n            if np.sum(y_true_ss[:, i]) > 0 and len(np.unique(y_true_ss[:, i])) > 1:\n                birdnet_auc_dict[sp] = roc_auc_score(y_true_ss[:, i], y_pred_birdnet_ss[:, i])\n                \n        birdnet_macro_auc = np.mean(list(birdnet_auc_dict.values())) if birdnet_auc_dict else np.nan\n        print(f\"BirdNET Zero-Shot Soundscape Macro AUC: {birdnet_macro_auc:.4f}\")\n        \n        # Save Results\n        summary_df = pd.DataFrame([{\n            'system': 'Perch Prototype',\n            'soundscape_macro_auc': perch_macro_auc\n        }, {\n            'system': 'BirdNET Zero-Shot',\n            'soundscape_macro_auc': birdnet_macro_auc\n        }])\n        summary_df.to_csv(REPORTS_OUT / 'soundscape_results.csv', index=False)\n        display(summary_df)\n    else:\n        print(\"No train parquets found. Cannot build prototypes for evaluation.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Species Bucket Analytics ---\nprint(\"\\n--- Species Bucket Analytics (Rarity Breakdown) ---\")\n\n# 1. Get training counts (samples per species)\ntrain_counts = train_perch_df['species_code'].value_counts().to_dict()\n\n# 2. Define buckets based on training abundance\nbuckets = [\n    ('<10 (Very Rare)', 0, 10),\n    ('10-50 (Rare)', 10, 50),\n    ('50-200 (Common)', 50, 200),\n    ('200+ (Very Common)', 200, float('inf'))\n]\n\nbucket_stats = []\nfor label, low, high in buckets:\n    # Get species in this bucket that have AUC values\n    perch_vals = [perch_auc_dict[sp] for sp in valid_ss_species if sp in perch_auc_dict and low <= train_counts.get(sp, 0) < high]\n    birdnet_vals = [birdnet_auc_dict[sp] for sp in valid_ss_species if sp in birdnet_auc_dict and low <= train_counts.get(sp, 0) < high]\n    \n    bucket_stats.append({\n        'Bucket': label,\n        'N Species': len(perch_vals),\n        'Perch AUC': np.mean(perch_vals) if perch_vals else np.nan,\n        'BirdNET AUC': np.mean(birdnet_vals) if birdnet_vals else np.nan\n    })\n\nbucket_df = pd.DataFrame(bucket_stats)\ndisplay(bucket_df)\n\n# 3. Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(10, 6))\nmelted_df = bucket_df.melt(id_vars='Bucket', value_vars=['Perch AUC', 'BirdNET AUC'], var_name='Model', value_name='AUC')\nsns.barplot(data=melted_df, x='Bucket', y='AUC', hue='Model', palette='viridis')\nplt.title('Soundscape Performance by Species Training Abundance')\nplt.ylim(0.5, 1.0)\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}