{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"datasetVersion","sourceId":8209908,"datasetId":4865209,"databundleVersionId":8334538},{"sourceType":"datasetVersion","sourceId":8682954,"datasetId":4783443,"databundleVersionId":8834739},{"sourceType":"datasetVersion","sourceId":8665785,"datasetId":4966492,"databundleVersionId":8816811},{"sourceType":"datasetVersion","sourceId":8748557,"datasetId":5213112,"databundleVersionId":8903409},{"sourceType":"modelInstanceVersion","sourceId":32637,"databundleVersionId":8261530,"modelInstanceId":26649,"modelId":37756},{"sourceType":"modelInstanceVersion","sourceId":516989,"databundleVersionId":13353982,"modelInstanceId":404337,"modelId":319}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Day 4: Mega-Ensemble Benchmark (Winners + Perch + BirdNET)\n\nThis notebook integrates the 1st Place (Team Cerberus) and 3rd Place (TheoViel) winning solutions for BirdCLEF 2024, along with the Perch and BirdNET baseline models. We evaluate the \"Ultimate Ensemble\" on our local 100-labeled soundscape benchmark.","metadata":{}},{"cell_type":"code","source":"# Mandatory setup for Perch and BirdNET to work correctly on Kaggle\n!pip install -q --upgrade \"tensorflow[and-cuda]>=2.16.1\" birdnetlib pyarrow scikit-learn resampy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:24:27.176950Z","iopub.execute_input":"2026-04-26T11:24:27.177296Z","iopub.status.idle":"2026-04-26T11:26:03.298318Z","shell.execute_reply.started":"2026-04-26T11:24:27.177272Z","shell.execute_reply":"2026-04-26T11:26:03.297568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:26:03.299934Z","iopub.execute_input":"2026-04-26T11:26:03.300367Z","iopub.status.idle":"2026-04-26T11:26:22.679867Z","shell.execute_reply.started":"2026-04-26T11:26:03.300333Z","shell.execute_reply":"2026-04-26T11:26:22.679215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport torch\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport tensorflow as tf\nimport timm\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom sklearn.metrics import roc_auc_score\n\n\n# --- 1. GLOBAL PATHS ---\nINPUT_DIR = Path('/kaggle/input')\n\ndef find_path(name):\n    try:\n        return next(INPUT_DIR.rglob(name))\n    except StopIteration:\n        return None\n\n# Find Everything Automatically\nmetadata_path = find_path('train_metadata.csv')\nCOMP_DIR = metadata_path.parent if metadata_path else None\nEVAL_DIR = find_path('labeled-soundscapes-birdclef-2024') or find_path('100-labeled-soundscapes-for-birdclef-2024')\nperch_path = str(find_path('saved_model.pb').parent)\nsrc_path = find_path('model_zoo')\nif src_path: sys.path.append(str(src_path.parent))\n\n# Dynamic Weight Folders\nTHEO_WEIGHTS_DIR = find_path('efficientvit_b0_0.pt').parent if find_path('efficientvit_b0_0.pt') else None\nFIRST_PLACE_DIR = find_path('effnet_seg20_80low.ckpt').parent if find_path('effnet_seg20_80low.ckpt') else None\n\nprint(f\"Metadata: {metadata_path}\\nWeights (1st): {FIRST_PLACE_DIR}\\nWeights (3rd): {THEO_WEIGHTS_DIR}\")\n\n# --- 2. IMPORTS ---\nfrom model_zoo.models import define_model\nfrom util.logger import Config\nfrom birdnetlib.analyzer import Analyzer\nfrom birdnetlib import Recording\n\n# --- 3. INITIALIZATION ---\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ntrain_metadata = pd.read_csv(metadata_path)\nTARGET_SPECIES = sorted(train_metadata['primary_label'].unique())\n\nperch_model = tf.saved_model.load(perch_path)\nperch_infer = perch_model.signatures[\"serving_default\"]\n\nprint(\"\\n--- ALL SYSTEMS INITIALIZED ---\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:26:22.680903Z","iopub.execute_input":"2026-04-26T11:26:22.681503Z","iopub.status.idle":"2026-04-26T11:29:28.204524Z","shell.execute_reply.started":"2026-04-26T11:26:22.681476Z","shell.execute_reply":"2026-04-26T11:29:28.203854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Model Loading (Winners + Baselines)","metadata":{}},{"cell_type":"code","source":"import torch\nimport timm\nimport json\nfrom birdnetlib.analyzer import Analyzer\n# --- 1. THEOVIEL (3rd Place) ---\ntheo_models = []\ntheo_v_pth = find_path('efficientvit_b0_0.pt')\nif theo_v_pth:\n    print(\"Building TheoViel Models...\")\n    THEO_DIR = theo_v_pth.parent\n    with open(THEO_DIR / 'config.json', 'r') as f:\n        config_dict = json.load(f)\n        config_dict['melspec_config'].pop('n_channels', None)\n        config = Config(config_dict)\n    for fold in [0, 1]:\n        model = define_model(config.name, config.melspec_config, head=config.head, \n                             num_classes=config.num_classes, verbose=False, \n                             pretrained=False, n_channels=3)\n        state_dict = torch.load(THEO_DIR / f'efficientvit_b0_{fold}.pt', map_location='cpu')\n        model.load_state_dict(state_dict['model'] if 'model' in state_dict else state_dict)\n        theo_models.append((model.to(DEVICE).eval(), config))\n        print(f\"  Theo Fold {fold} Loaded.\")\n# --- 2. CERBERUS (1st Place) ---\nfirst_place_models = []\ncerb_v_pth = find_path('effnet_seg20_80low.ckpt')\nif cerb_v_pth:\n    print(\"\\nBuilding Cerberus Models...\")\n    CERB_DIR = cerb_v_pth.parent\n    for name in ['effnet_seg20_80low.ckpt', 'regnety_seg60_fold0.ckpt']:\n        state_dict = torch.load(CERB_DIR / name, map_location='cpu')\n        m_name = 'efficientnet_b0' if 'effnet' in name else 'regnety_008.pycls_in1k'\n        model = timm.create_model(m_name, pretrained=None, num_classes=182, in_chans=1)\n        model.load_state_dict({k[6:]: v for k, v in state_dict['state_dict'].items() if k.startswith('model.')})\n        first_place_models.append(model.to(DEVICE).eval())\n        print(f\"  Cerberus {name} Loaded.\")\n# --- 3. PERCH & BIRDNET ---\nprint(\"\\nInitializing Foundation Models...\")\nperch_path = str(find_path('saved_model.pb').parent)\nperch_model = tf.saved_model.load(perch_path)\nperch_infer = perch_model.signatures[\"serving_default\"]\n# This is what was missing!\nbirdnet_analyzer = Analyzer()\nprint(f\"\\nALL SYSTEMS READY: Theo({len(theo_models)}), Cerberus({len(first_place_models)}), Perch, BirdNET\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:31:06.188178Z","iopub.execute_input":"2026-04-26T11:31:06.188968Z","iopub.status.idle":"2026-04-26T11:32:01.737010Z","shell.execute_reply.started":"2026-04-26T11:31:06.188937Z","shell.execute_reply":"2026-04-26T11:32:01.736253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Load the REAL Perch labels we just found\nperch_labels_path = '/kaggle/input/models/google/bird-vocalization-classifier/tensorflow2/perch_v2/2/assets/perch_v2_ebird_classes.csv'\nperch_labels_df = pd.read_csv(perch_labels_path)\n# 2. Map the 182 Competition birds to their Perch indices\nperch_idx_map = []\nfor species in TARGET_SPECIES:\n    try:\n        # Match the competition species code to the 'ebird2021' column in Perch\n        idx = perch_labels_df[perch_labels_df['ebird2021'] == species].index[0]\n        perch_idx_map.append(idx)\n    except (IndexError, KeyError):\n        # If a bird is missing from Perch, use -1\n        perch_idx_map.append(-1)\nprint(f\"Successfully mapped {len([i for i in perch_idx_map if i != -1])} out of 182 species.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:32:16.502122Z","iopub.execute_input":"2026-04-26T11:32:16.502852Z","iopub.status.idle":"2026-04-26T11:32:16.747924Z","shell.execute_reply.started":"2026-04-26T11:32:16.502819Z","shell.execute_reply":"2026-04-26T11:32:16.747103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. BirdNET Mapping ---\n# Map BirdNET Scientific names to our Species Codes\nbn_labels = birdnet_analyzer.labels\nbirdnet_idx_map = []\nfor species in TARGET_SPECIES:\n    # Get scientific name from competition metadata\n    sci_name = train_metadata[train_metadata['primary_label'] == species]['scientific_name'].iloc[0]\n    try:\n        # Find index where the scientific name matches\n        idx = [i for i, label in enumerate(bn_labels) if sci_name.lower() in label.lower()][0]\n        birdnet_idx_map.append(idx)\n    except IndexError:\n        birdnet_idx_map.append(-1)\nprint(f\"Mapped Perch: {len([i for i in perch_idx_map if i != -1])}/182\")\nprint(f\"Mapped BirdNET: {len([i for i in birdnet_idx_map if i != -1])}/182\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:32:19.416701Z","iopub.execute_input":"2026-04-26T11:32:19.417306Z","iopub.status.idle":"2026-04-26T11:32:20.159417Z","shell.execute_reply.started":"2026-04-26T11:32:19.417273Z","shell.execute_reply":"2026-04-26T11:32:20.158565Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Inference Loop","metadata":{}},{"cell_type":"code","source":"# --- SELF-CONTAINED MEGA-INFERENCE ---\nimport numpy as np\nimport torch\nimport librosa\nfrom tqdm.auto import tqdm\nimport pandas as pd\n# 1. Verify Audio Files\nEVAL_DIR = Path('/kaggle/input/datasets/richolson/100-labeled-soundscapes-for-birdclef-2024/')\naudio_files = sorted(list(EVAL_DIR.glob(\"*.ogg\")))\nprint(f\"Verified: Found {len(audio_files)} files to process.\")\nall_preds = []\nstep_size, chunk_size = int(32000 * 2.5), int(32000 * 5.0)\nfor audio_p in tqdm(audio_files):\n    try:\n        y, sr = librosa.load(audio_p, sr=32000)\n        \n        # 1st Place specific MelSpec (Calculated once per file)\n        melspec_1st = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128, fmin=40, fmax=15000)\n        melspec_1st = librosa.power_to_db(melspec_1st).astype(np.float32)\n        for start in range(0, len(y) - chunk_size + 1, step_size):\n            chunk = y[start : start + chunk_size]\n            chunk_torch = torch.from_numpy(chunk[None, :]).to(DEVICE)\n            ensemble_scores = []\n            \n            # A. Perch\n            p_raw = perch_infer(inputs=tf.constant(chunk[None, :], dtype=tf.float32))['label'].numpy()[0]\n            ensemble_scores.append(np.array([p_raw[i] if i != -1 else 0.0 for i in perch_idx_map]))\n            \n    # --- Update this part in your Inference Loop ---\n            # B. Theo's Models (3rd Place)\n            with torch.no_grad():\n                for m, cfg in theo_models:\n                    # We add [0] here to extract the logits from the tuple\n                    out = m(chunk_torch)\n                    if isinstance(out, tuple):\n                        out = out[0]\n                    ensemble_scores.append(torch.sigmoid(out).cpu().numpy()[0])\n            \n            # C. Cerberus (1st Place)\n            m_start = int((start/sr) * (sr/512))\n            m_chunk = melspec_1st[:, m_start : m_start + 313]\n            m_chunk_torch = torch.from_numpy(m_chunk[None, None, :, :]).to(DEVICE)\n            with torch.no_grad():\n                for m in first_place_models:\n                    ensemble_scores.append(torch.sigmoid(m(m_chunk_torch)).cpu().numpy()[0])\n            \n            # Final Average\n            all_preds.append({\n                'audio_id': audio_p.stem, \n                'time': (start+chunk_size)/sr, \n                'pred': np.mean(ensemble_scores, axis=0)\n            })\n    except Exception as e:\n        print(f\"Error processing {audio_p.name}: {e}\")\n# Save to results_df immediately so it's ready for scoring\nresults_df = pd.DataFrame(all_preds)\nprint(f\"Success! Processed {len(results_df)} chunks.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:34:30.701755Z","iopub.execute_input":"2026-04-26T11:34:30.702413Z","iopub.status.idle":"2026-04-26T11:43:50.280858Z","shell.execute_reply.started":"2026-04-26T11:34:30.702381Z","shell.execute_reply":"2026-04-26T11:43:50.280106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import roc_auc_score\n# 1. Load labels\ngt_df = pd.read_csv('/kaggle/input/datasets/richolson/100-labeled-soundscapes-for-birdclef-2024/labeled_soundscapes.csv')\n# 2. Align results\nresults_df = pd.DataFrame(all_preds)\nresults_df['row_id'] = results_df['audio_id'].astype(str) + '_' + results_df['time'].astype(int).astype(str)\nmerged = gt_df.merge(results_df, on='row_id')\n# 3. Extract matrices and force size 182\ny_true = merged[TARGET_SPECIES].values[:, :182]\ny_pred = np.stack(merged['pred'].values)[:, :182]\n# 4. Calculate per-species AUC\nvalid_indices = [i for i in range(y_true.shape[1]) if 0 < y_true[:, i].sum() < y_true.shape[0]]\nspecies_scores = []\nfor i in valid_indices:\n    score = roc_auc_score(y_true[:, i], y_pred[:, i])\n    species_scores.append({'species': TARGET_SPECIES[i], 'auc': score})\nfinal_scores = pd.DataFrame(species_scores).sort_values('auc', ascending=False)\n# 5. Output\nmacro_auc = final_scores['auc'].mean()\nprint(f\"ULTIMATE MEGA-ENSEMBLE MACRO AUC: {macro_auc:.4f}\")\nprint(f\"Calculated on {len(final_scores)} valid species.\")\nprint(\"\\nTop 10 Performers:\")\nprint(final_scores.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:44:01.854183Z","iopub.execute_input":"2026-04-26T11:44:01.854753Z","iopub.status.idle":"2026-04-26T11:44:02.331158Z","shell.execute_reply.started":"2026-04-26T11:44:01.854711Z","shell.execute_reply":"2026-04-26T11:44:02.330470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(final_scores.tail(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:44:06.541953Z","iopub.execute_input":"2026-04-26T11:44:06.542647Z","iopub.status.idle":"2026-04-26T11:44:06.548647Z","shell.execute_reply.started":"2026-04-26T11:44:06.542615Z","shell.execute_reply":"2026-04-26T11:44:06.547868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the species breakdown (to see which birds were hardest)\nfinal_scores.to_csv('mega_ensemble_per_species_auc.csv', index=False)\n# Save the full predictions (so you don't have to rerun the loop)\nresults_df.to_csv('mega_ensemble_predictions.csv', index=False)\nprint(\"Results saved to mega_ensemble_per_species_auc.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:48:37.413483Z","iopub.execute_input":"2026-04-26T11:48:37.414262Z","iopub.status.idle":"2026-04-26T11:48:45.290766Z","shell.execute_reply.started":"2026-04-26T11:48:37.414229Z","shell.execute_reply":"2026-04-26T11:48:45.290083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Merge AUC with training counts\ntrain_counts = train_metadata['primary_label'].value_counts().reset_index()\ntrain_counts.columns = ['species', 'count']\nplot_df = final_scores.merge(train_counts, on='species')\n\nplt.figure(figsize=(10, 6))\nsns.regplot(data=plot_df, x='count', y='auc', scatter_kws={'alpha':0.5}, line_kws={'color':'red'})\nplt.xscale('log')\nplt.title('The Long-Tail Problem: Training Samples vs. AUC')\nplt.xlabel('Number of Training Samples (Log Scale)')\nplt.ylabel('Macro AUC on Soundscapes')\nplt.grid(True, which=\"both\", ls=\"-\", alpha=0.2)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:49:15.817742Z","iopub.execute_input":"2026-04-26T11:49:15.818203Z","iopub.status.idle":"2026-04-26T11:49:16.496649Z","shell.execute_reply.started":"2026-04-26T11:49:15.818170Z","shell.execute_reply":"2026-04-26T11:49:16.495975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming you have your Day 2 Focal AUC handy (e.g., 0.98)\ndata = {\n    'Dataset': ['Clean (Focal)', 'Real-World (Soundscape)'],\n    'AUC': [0.985, macro_auc] # Replace 0.985 with your actual Focal AUC\n}\nshift_df = pd.DataFrame(data)\nplt.figure(figsize=(8, 6))\nsns.barplot(data=shift_df, x='Dataset', y='AUC', palette='viridis')\nplt.ylim(0.8, 1.0)\nplt.title('The Domain Shift Gap (Focal vs. Soundscape)')\nplt.ylabel('Macro AUC')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:49:52.337526Z","iopub.execute_input":"2026-04-26T11:49:52.338132Z","iopub.status.idle":"2026-04-26T11:49:52.474128Z","shell.execute_reply.started":"2026-04-26T11:49:52.338101Z","shell.execute_reply":"2026-04-26T11:49:52.473323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.hist(final_scores['auc'], bins=20, cumulative=True, density=True, histtype='step', linewidth=2)\nplt.axvline(0.9, color='red', linestyle='--', label='90% AUC Threshold')\nplt.title('Cumulative Difficulty Distribution')\nplt.xlabel('AUC Score')\nplt.ylabel('Fraction of Species')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:50:29.201997Z","iopub.execute_input":"2026-04-26T11:50:29.202561Z","iopub.status.idle":"2026-04-26T11:50:29.336952Z","shell.execute_reply.started":"2026-04-26T11:50:29.202532Z","shell.execute_reply":"2026-04-26T11:50:29.336353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find birds with many training samples (>100) but low AUC (<0.8)\ndomain_shift_victims = plot_df[(plot_df['count'] > 100) & (plot_df['auc'] < 0.8)]\nprint(\"Birds that are common in training but failing in the wild:\")\nprint(domain_shift_victims[['species', 'count', 'auc']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:55:39.327609Z","iopub.execute_input":"2026-04-26T11:55:39.328420Z","iopub.status.idle":"2026-04-26T11:55:39.336086Z","shell.execute_reply.started":"2026-04-26T11:55:39.328387Z","shell.execute_reply":"2026-04-26T11:55:39.335391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Categorize birds by their training sample count\ndef get_bucket(count):\n    if count < 20: return 'Very Rare (<20)'\n    if count < 100: return 'Rare (20-100)'\n    return 'Common (>100)'\nplot_df['bucket'] = plot_df['count'].apply(get_bucket)\n# 2. Calculate Average AUC for each bucket\nbucket_results = plot_df.groupby('bucket')['auc'].mean().reindex(['Common (>100)', 'Rare (20-100)', 'Very Rare (<20)'])\n# 3. Plot the problem\nplt.figure(figsize=(10, 6))\nsns.barplot(x=bucket_results.index, y=bucket_results.values, palette='Reds_r')\nplt.ylim(0.8, 1.0)\nplt.title('The Long-Tail Problem: AUC by Species Rarity')\nplt.ylabel('Average Macro AUC')\n# Replace the last few lines with this:\nfor i, v in enumerate(bucket_results.values):\n    plt.text(i, v + 0.005, f\"{v:.4f}\", ha='center', weight='bold') # Changed fontWeight to weight\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T11:57:13.212996Z","iopub.execute_input":"2026-04-26T11:57:13.213556Z","iopub.status.idle":"2026-04-26T11:57:13.410502Z","shell.execute_reply.started":"2026-04-26T11:57:13.213520Z","shell.execute_reply":"2026-04-26T11:57:13.409368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Trying Power-mean-ensemble","metadata":{}},{"cell_type":"code","source":"# --- SELF-CONTAINED MEGA-INFERENCE ---\nimport numpy as np\nimport torch\nimport librosa\nfrom tqdm.auto import tqdm\nimport pandas as pd\n# 1. Verify Audio Files\nEVAL_DIR = Path('/kaggle/input/datasets/richolson/100-labeled-soundscapes-for-birdclef-2024/')\naudio_files = sorted(list(EVAL_DIR.glob(\"*.ogg\")))\nprint(f\"Verified: Found {len(audio_files)} files to process.\")\nall_preds = []\nstep_size, chunk_size = int(32000 * 2.5), int(32000 * 5.0)\nfor audio_p in tqdm(audio_files):\n    try:\n        y, sr = librosa.load(audio_p, sr=32000)\n        \n        # 1st Place specific MelSpec (Calculated once per file)\n        melspec_1st = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128, fmin=40, fmax=15000)\n        melspec_1st = librosa.power_to_db(melspec_1st).astype(np.float32)\n        for start in range(0, len(y) - chunk_size + 1, step_size):\n            chunk = y[start : start + chunk_size]\n            chunk_torch = torch.from_numpy(chunk[None, :]).to(DEVICE)\n            ensemble_scores = []\n            \n            # A. Perch\n            p_raw = perch_infer(inputs=tf.constant(chunk[None, :], dtype=tf.float32))['label'].numpy()[0]\n            ensemble_scores.append(np.array([p_raw[i] if i != -1 else 0.0 for i in perch_idx_map]))\n            \n    # --- Update this part in your Inference Loop ---\n            # B. Theo's Models (3rd Place)\n            with torch.no_grad():\n                for m, cfg in theo_models:\n                    # We add [0] here to extract the logits from the tuple\n                    out = m(chunk_torch)\n                    if isinstance(out, tuple):\n                        out = out[0]\n                    ensemble_scores.append(torch.sigmoid(out).cpu().numpy()[0])\n            \n            # C. Cerberus (1st Place)\n            m_start = int((start/sr) * (sr/512))\n            m_chunk = melspec_1st[:, m_start : m_start + 313]\n            m_chunk_torch = torch.from_numpy(m_chunk[None, None, :, :]).to(DEVICE)\n            with torch.no_grad():\n                for m in first_place_models:\n                    ensemble_scores.append(torch.sigmoid(m(m_chunk_torch)).cpu().numpy()[0])\n            \n    # --- NEW: POWER-MEAN ENSEMBLE (Confidence Weighted) ---\n            # Squaring the scores rewards high-confidence and suppresses low-level noise\n            power_scores = [s**2 for s in ensemble_scores]\n            final_pred = np.sqrt(np.mean(power_scores, axis=0))\n            \n            all_preds.append({\n                'audio_id': audio_p.stem, \n                'time': (start+chunk_size)/sr, \n                'pred': final_pred\n            })\n    except Exception as e:\n        print(f\"Error processing {audio_p.name}: {e}\")\n# Save to results_df immediately so it's ready for scoring\nresults_df = pd.DataFrame(all_preds)\nprint(f\"Success! Processed {len(results_df)} chunks.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T12:02:44.922958Z","iopub.execute_input":"2026-04-26T12:02:44.923565Z","iopub.status.idle":"2026-04-26T12:12:04.300161Z","shell.execute_reply.started":"2026-04-26T12:02:44.923534Z","shell.execute_reply":"2026-04-26T12:12:04.299326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.metrics import roc_auc_score\n# 1. Load labels\ngt_df = pd.read_csv('/kaggle/input/datasets/richolson/100-labeled-soundscapes-for-birdclef-2024/labeled_soundscapes.csv')\n# 2. Align results\nresults_df = pd.DataFrame(all_preds)\nresults_df['row_id'] = results_df['audio_id'].astype(str) + '_' + results_df['time'].astype(int).astype(str)\nmerged = gt_df.merge(results_df, on='row_id')\n# 3. Extract matrices and force size 182\ny_true = merged[TARGET_SPECIES].values[:, :182]\ny_pred = np.stack(merged['pred'].values)[:, :182]\n# 4. Calculate per-species AUC\nvalid_indices = [i for i in range(y_true.shape[1]) if 0 < y_true[:, i].sum() < y_true.shape[0]]\nspecies_scores = []\nfor i in valid_indices:\n    score = roc_auc_score(y_true[:, i], y_pred[:, i])\n    species_scores.append({'species': TARGET_SPECIES[i], 'auc': score})\nfinal_scores = pd.DataFrame(species_scores).sort_values('auc', ascending=False)\n# 5. Output\nmacro_auc = final_scores['auc'].mean()\nprint(f\"ULTIMATE MEGA-ENSEMBLE MACRO AUC: {macro_auc:.4f}\")\nprint(f\"Calculated on {len(final_scores)} valid species.\")\nprint(\"\\nTop 10 Performers:\")\nprint(final_scores.head(10))\nprint(final_scores.tail(10))\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Merge AUC with training counts\ntrain_counts = train_metadata['primary_label'].value_counts().reset_index()\ntrain_counts.columns = ['species', 'count']\nplot_df = final_scores.merge(train_counts, on='species')\n\nplt.figure(figsize=(10, 6))\nsns.regplot(data=plot_df, x='count', y='auc', scatter_kws={'alpha':0.5}, line_kws={'color':'red'})\nplt.xscale('log')\nplt.title('The Long-Tail Problem: Training Samples vs. AUC')\nplt.xlabel('Number of Training Samples (Log Scale)')\nplt.ylabel('Macro AUC on Soundscapes')\nplt.grid(True, which=\"both\", ls=\"-\", alpha=0.2)\nplt.show()\n# Assuming you have your Day 2 Focal AUC handy (e.g., 0.98)\ndata = {\n    'Dataset': ['Clean (Focal)', 'Real-World (Soundscape)'],\n    'AUC': [0.985, macro_auc] # Replace 0.985 with your actual Focal AUC\n}\nshift_df = pd.DataFrame(data)\nplt.figure(figsize=(8, 6))\nsns.barplot(data=shift_df, x='Dataset', y='AUC', palette='viridis')\nplt.ylim(0.8, 1.0)\nplt.title('The Domain Shift Gap (Focal vs. Soundscape)')\nplt.ylabel('Macro AUC')\nplt.show()\nplt.figure(figsize=(10, 6))\nplt.hist(final_scores['auc'], bins=20, cumulative=True, density=True, histtype='step', linewidth=2)\nplt.axvline(0.9, color='red', linestyle='--', label='90% AUC Threshold')\nplt.title('Cumulative Difficulty Distribution')\nplt.xlabel('AUC Score')\nplt.ylabel('Fraction of Species')\nplt.legend()\nplt.show()\n# Find birds with many training samples (>100) but low AUC (<0.8)\ndomain_shift_victims = plot_df[(plot_df['count'] > 100) & (plot_df['auc'] < 0.8)]\nprint(\"Birds that are common in training but failing in the wild:\")\nprint(domain_shift_victims[['species', 'count', 'auc']])\n# 1. Categorize birds by their training sample count\ndef get_bucket(count):\n    if count < 20: return 'Very Rare (<20)'\n    if count < 100: return 'Rare (20-100)'\n    return 'Common (>100)'\nplot_df['bucket'] = plot_df['count'].apply(get_bucket)\n# 2. Calculate Average AUC for each bucket\nbucket_results = plot_df.groupby('bucket')['auc'].mean().reindex(['Common (>100)', 'Rare (20-100)', 'Very Rare (<20)'])\n# 3. Plot the problem\nplt.figure(figsize=(10, 6))\nsns.barplot(x=bucket_results.index, y=bucket_results.values, palette='Reds_r')\nplt.ylim(0.8, 1.0)\nplt.title('The Long-Tail Problem: AUC by Species Rarity')\nplt.ylabel('Average Macro AUC')\n# Replace the last few lines with this:\nfor i, v in enumerate(bucket_results.values):\n    plt.text(i, v + 0.005, f\"{v:.4f}\", ha='center', weight='bold') # Changed fontWeight to weight\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T12:12:13.211013Z","iopub.execute_input":"2026-04-26T12:12:13.211405Z","iopub.status.idle":"2026-04-26T12:12:14.354422Z","shell.execute_reply.started":"2026-04-26T12:12:13.211379Z","shell.execute_reply":"2026-04-26T12:12:14.353549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check how many birds are labeled in each 5-second row\nrow_sums = gt_df[TARGET_SPECIES].sum(axis=1)\nprint(f\"Max birds in a single 5s chunk: {row_sums.max()}\")\nprint(f\"Number of chunks with 0 birds: {(row_sums == 0).sum()}\")\nprint(f\"Number of chunks with 1 bird:  {(row_sums == 1).sum()}\")\nprint(f\"Number of chunks with 2+ birds: {(row_sums >= 2).sum()}\")\n# Show a row with multiple birds if it exists\nif row_sums.max() > 1:\n    multi_bird_row = gt_df[row_sums > 1].head(1)\n    print(\"\\nExample of a multi-bird row:\")\n    # Show only the columns that have a 1\n    print(multi_bird_row.columns[(multi_bird_row == 1).iloc[0]].tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T12:23:39.652940Z","iopub.execute_input":"2026-04-26T12:23:39.653722Z","iopub.status.idle":"2026-04-26T12:23:39.667272Z","shell.execute_reply.started":"2026-04-26T12:23:39.653689Z","shell.execute_reply":"2026-04-26T12:23:39.666324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the species, their counts, their scores, and their rarity buckets\nplot_df[['species', 'count', 'auc', 'bucket']].to_csv('species_rarity_buckets.csv', index=False)\nprint(\"Bucket list saved to species_rarity_buckets.csv\")\nprint(\"\\nBucket Counts:\")\nprint(plot_df['bucket'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T12:27:13.357867Z","iopub.execute_input":"2026-04-26T12:27:13.358198Z","iopub.status.idle":"2026-04-26T12:27:13.367109Z","shell.execute_reply.started":"2026-04-26T12:27:13.358170Z","shell.execute_reply":"2026-04-26T12:27:13.366381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}