{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"datasetVersion","sourceId":8209908,"datasetId":4865209,"databundleVersionId":8334538},{"sourceType":"kernelVersion","sourceId":314625876},{"sourceType":"kernelVersion","sourceId":314648639},{"sourceType":"kernelVersion","sourceId":314956547}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2024: Day 7 - Scientific Ablation Study\nThis notebook isolates the impact of each of our '16 micro-upgrades' by retraining the head multiple times with specific features disabled. This transforms engineering tweaks into scientific findings.","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom sklearn.metrics import roc_auc_score\nimport time\n\nINPUT_DIR = Path('/kaggle/input')\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nEPOCHS = 30 # Reduced epochs for faster ablation runs\nBATCH_SIZE = 256\nLR = 1e-3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:18:08.745953Z","iopub.execute_input":"2026-04-28T19:18:08.746686Z","iopub.status.idle":"2026-04-28T19:18:14.949808Z","shell.execute_reply.started":"2026-04-28T19:18:08.746652Z","shell.execute_reply":"2026-04-28T19:18:14.948707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Modular Architecture & Loss","metadata":{}},{"cell_type":"code","source":"class MLPHead(nn.Module):\n    def __init__(self, in_dim=1536, hidden_dim=512, n_classes=182, use_residual=True):\n        super().__init__()\n        self.use_residual = use_residual\n        self.input_layer = nn.Sequential(nn.Linear(in_dim, hidden_dim), nn.LayerNorm(hidden_dim), nn.GELU())\n        self.res_block = nn.Sequential(nn.Linear(hidden_dim, hidden_dim), nn.LayerNorm(hidden_dim), nn.GELU())\n        self.output_layer = nn.Linear(hidden_dim, n_classes)\n        self.temperature = nn.Parameter(torch.ones(n_classes))\n        \n    def forward(self, x, apply_calib=False):\n        x = self.input_layer(x)\n        if self.use_residual:\n            x = x + self.res_block(x)\n        else:\n            x = self.res_block(x)\n        logits = self.output_layer(x)\n        if apply_calib:\n            logits = logits / (F.softplus(self.temperature) + 1e-4)\n        return logits\n\ndef logit_adjusted_loss(logits, targets, log_prior, tau=1.0):\n    # Standard Focal Loss implementation\n    logits_adj = logits + tau * log_prior\n    bce = F.binary_cross_entropy_with_logits(logits_adj, targets, reduction='none')\n    p = torch.sigmoid(logits_adj)\n    p_t = p * targets + (1 - p) * (1 - targets)\n    focal_weight = (1 - p_t) ** 2.0\n    return (focal_weight * bce).mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:18:14.951440Z","iopub.execute_input":"2026-04-28T19:18:14.951989Z","iopub.status.idle":"2026-04-28T19:18:14.963273Z","shell.execute_reply.started":"2026-04-28T19:18:14.951943Z","shell.execute_reply":"2026-04-28T19:18:14.962154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MixupEmbeddingDataset(Dataset):\n    def __init__(self, f_emb, f_y, s_emb=None, s_y=None, p_mix=0.5):\n        self.f_emb = torch.tensor(f_emb, dtype=torch.float32)\n        self.f_y = torch.tensor(f_y, dtype=torch.float32)\n        self.s_emb = torch.tensor(s_emb, dtype=torch.float32) if s_emb is not None else None\n        self.s_y = torch.tensor(s_y, dtype=torch.float32) if s_y is not None else None\n        self.p_mix = p_mix # Store the probability here\n        \n    def __len__(self): return len(self.f_emb)\n    \n    def __getitem__(self, idx):\n        x, y = self.f_emb[idx], self.f_y[idx]\n        \n        # Use p_mix to decide whether to apply mixup\n        if self.s_emb is not None and len(self.s_emb) > 0 and np.random.rand() < self.p_mix:\n            j = np.random.randint(0, len(self.s_emb))\n            lam = np.random.beta(0.4, 0.4)\n            x = lam * x + (1 - lam) * self.s_emb[j]\n            y = lam * y + (1 - lam) * self.s_y[j]\n            \n        return x, y\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:18:14.964407Z","iopub.execute_input":"2026-04-28T19:18:14.964842Z","iopub.status.idle":"2026-04-28T19:18:14.987596Z","shell.execute_reply.started":"2026-04-28T19:18:14.964787Z","shell.execute_reply":"2026-04-28T19:18:14.986566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Global Data Loading","metadata":{}},{"cell_type":"code","source":"# Load 182 species master list\nmeta_df = pd.read_csv(INPUT_DIR / 'competitions/birdclef-2024/train_metadata.csv')\nALL_SPECIES = sorted(meta_df['primary_label'].unique())\nN_CLASSES = len(ALL_SPECIES)\nsp_to_idx = {sp: i for i, sp in enumerate(ALL_SPECIES)}\n\n# Load Focal Embeddings\ntrain_files = list(INPUT_DIR.rglob('perch_train*.parquet'))\ntrain_df = pd.concat([pd.read_parquet(f) for f in train_files]).reset_index(drop=True)\nfull_focal_emb = np.stack(train_df['emb'].values)\nfull_focal_y = np.zeros((len(full_focal_emb), N_CLASSES))\nfor i, sp in enumerate(train_df['species_code']): \n    if sp in sp_to_idx: full_focal_y[i, sp_to_idx[sp]] = 1\n\n# Split for internal validation (80/20)\nidx = np.random.permutation(len(full_focal_emb))\nsplit = int(0.8 * len(idx))\ntrain_idx, val_idx = idx[:split], idx[split:]\n\nfocal_train_emb, focal_train_y = full_focal_emb[train_idx], full_focal_y[train_idx]\nfocal_val_emb, focal_val_y = full_focal_emb[val_idx], full_focal_y[val_idx]\n\n# Load Pseudo-labels if present\ntry:\n    pseudo_csv = next(INPUT_DIR.rglob('pseudo_labels_v1.csv'))\n    pseudo_npy = next(INPUT_DIR.rglob('unlabeled_embeddings_v1.npy'))\n    sound_emb = np.load(pseudo_npy)\n    pdf = pd.read_csv(pseudo_csv)\n    # Handle pred string parsing robustly\n    def parse_pred(p):\n        cleaned = p.replace('[','').replace(']','').replace('\\n',' ').split()\n        return np.array([float(x) for x in cleaned])\n    sound_logits = np.stack([parse_pred(p) for p in pdf['pred'].values])\n    sound_y = 1 / (1 + np.exp(-sound_logits)) # Sigmoid to get soft labels\nexcept:\n    print(\"Pseudo-labels not found. Ablating PL effect will be null.\")\n    sound_emb, sound_y = None, None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:18:14.990373Z","iopub.execute_input":"2026-04-28T19:18:14.990705Z","iopub.status.idle":"2026-04-28T19:19:55.775011Z","shell.execute_reply.started":"2026-04-28T19:18:14.990673Z","shell.execute_reply":"2026-04-28T19:19:55.773962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Corrected Soundscape Load with ID Alignment ---\ntry:\n    ss_labels_path = next(Path('/kaggle/input').rglob('labeled_soundscapes.csv'))\n    ss_labels_df = pd.read_csv(ss_labels_path)\n    \n    ss_emb_path = next(Path('/kaggle/input').rglob('perch_soundscape.parquet'))\n    ss_emb_df = pd.read_parquet(ss_emb_path)\n    \n    # 1. Align the row_id format\n    # In the labels, it is 'audio_id_time'. Let's match that in the embeddings.\n    if 'time' in ss_emb_df.columns:\n        ss_emb_df['row_id'] = ss_emb_df['audio_id'].astype(str) + '_' + ss_emb_df['time'].astype(str)\n    else:\n        ss_emb_df['row_id'] = ss_emb_df['audio_id']\n\n    # 2. Merge\n    merged = ss_labels_df.merge(ss_emb_df, on='row_id')\n    \n    if len(merged) == 0:\n        # Debugging print to see what's wrong\n        print(f\"Sample Label ID: {ss_labels_df['row_id'].iloc[0]}\")\n        print(f\"Sample Embedding ID: {ss_emb_df['row_id'].iloc[0]}\")\n        raise ValueError(\"Merge resulted in 0 rows. Check ID formats above.\")\n\n    ss_emb = np.stack(merged['emb'].values)\n    # Use ALL_SPECIES from your master list to ensure column order is perfect\n    ss_y = merged[ALL_SPECIES].values\n    \n    print(f\"✅ SUCCESS: Aligned {len(ss_emb)} soundscape chunks.\")\n    \nexcept Exception as e:\n    print(f\"⚠️ Soundscape load failed: {e}\")\n    ss_emb, ss_y = np.zeros((0, 1536)), np.zeros((0, 182))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:20:53.447578Z","iopub.execute_input":"2026-04-28T19:20:53.448037Z","iopub.status.idle":"2026-04-28T19:20:53.757677Z","shell.execute_reply.started":"2026-04-28T19:20:53.448003Z","shell.execute_reply":"2026-04-28T19:20:53.756768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create index maps for buckets\nabundance = train_df['species_code'].value_counts()\nbucket_indices = {\n    'Very_Rare': [i for i, sp in enumerate(ALL_SPECIES) if abundance.get(sp, 0) < 10],\n    'Rare':      [i for i, sp in enumerate(ALL_SPECIES) if 10 <= abundance.get(sp, 0) < 50],\n    'Medium':    [i for i, sp in enumerate(ALL_SPECIES) if 50 <= abundance.get(sp, 0) < 200],\n    'Common':    [i for i, sp in enumerate(ALL_SPECIES) if abundance.get(sp, 0) >= 200]\n}\n\n# Example: Western Ghats Endemics (Subset based on competition metadata)\nwg_endemics = ['asbfly', 'ashdro1', 'ashpri1', 'ashwoo2', 'asikoe2', 'asiope1', 'aspfly1', 'aspswi1']\nwg_indices = [i for i, sp in enumerate(ALL_SPECIES) if sp in wg_endemics]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:29:21.280078Z","iopub.execute_input":"2026-04-28T19:29:21.280857Z","iopub.status.idle":"2026-04-28T19:29:21.292213Z","shell.execute_reply.started":"2026-04-28T19:29:21.280798Z","shell.execute_reply":"2026-04-28T19:29:21.291018Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. The Ablation Engine","metadata":{}},{"cell_type":"code","source":"def run_experiment(name, use_logit_adj=True, use_mixup=True, use_sampler=True, \n                   use_residual=True, use_pl=True, use_calib=True):\n    global focal_train_emb, focal_train_y, focal_val_emb, focal_val_y, sound_emb, sound_y, N_CLASSES, DEVICE, full_focal_y,ss_emb,ss_y\n    \n    print(f\"\\n🚀 Running: {name}\")\n    \n    # 1. Setup Dataset\n    s_emb = sound_emb if (use_pl and sound_emb is not None) else np.zeros((0, 1536))\n    s_y = sound_y if (use_pl and sound_y is not None) else np.zeros((0, N_CLASSES))\n    dataset = MixupEmbeddingDataset(focal_train_emb, focal_train_y, s_emb, s_y, p_mix=0.5 if use_mixup else 0.0)\n    \n    # 2. Setup Loader\n    if use_sampler:\n        freq = focal_train_y.sum(axis=0) + 1\n        sample_weights = (focal_train_y * (1.0 / freq)).sum(axis=1)\n        sampler = WeightedRandomSampler(sample_weights, len(sample_weights))\n        loader = DataLoader(dataset, batch_size=256, sampler=sampler)\n    else:\n        loader = DataLoader(dataset, batch_size=256, shuffle=True)\n        \n    # 3. Setup Model\n    model = MLPHead(in_dim=1536, n_classes=N_CLASSES, use_residual=use_residual).to(DEVICE)\n    \n    # 4. Setup Loss Logic\n    class_counts = full_focal_y.sum(axis=0) + 1\n    log_prior = torch.log(torch.tensor(class_counts / class_counts.sum())).to(DEVICE)\n    \n    # 5. Training Loop\n    optimizer = torch.optim.AdamW(model.parameters(), lr=1e-3)\n    for epoch in range(10):\n        model.train()\n        for x, y in loader:\n            x, y = x.to(DEVICE), y.to(DEVICE)\n            optimizer.zero_grad()\n            logits = model(x, apply_calib=False)\n            \n            # Apply Loss based on experiment\n            if use_logit_adj:\n                loss = logit_adjusted_loss(logits, y, log_prior)\n            else:\n                loss = nn.BCEWithLogitsLoss()(logits, y)\n                \n            loss.backward()\n            optimizer.step()\n            \n    # 6. Evaluate (NaN-Proof)\n    # --- Corrected Evaluation Block (Bottom of run_experiment) ---\n    # --- Scientific Evaluation Block ---\n        model.eval()\n        # --- Change these lines inside run_experiment ---\n        with torch.no_grad():\n            # Use ss_emb instead of focal_val_emb\n            v_preds = torch.sigmoid(model(torch.tensor(ss_emb).to(DEVICE))).cpu().numpy()\n            # Use ss_y instead of focal_val_y\n            v_targets = ss_y\n\n            \n        species_aucs = {}\n        for i in range(N_CLASSES):\n            if len(np.unique(v_targets[:, i])) > 1:\n                species_aucs[i] = roc_auc_score(v_targets[:, i], v_preds[:, i])\n                \n        # Calculate Bucket Metrics\n        report = {\"Experiment\": name}\n        for b_name, indices in bucket_indices.items():\n            b_aucs = [species_aucs[i] for i in indices if i in species_aucs]\n            report[b_name] = np.mean(b_aucs) if b_aucs else 0.0\n        \n        # Calculate Endemic Score\n        e_aucs = [species_aucs[i] for i in wg_indices if i in species_aucs]\n        report[\"Endemic_AUC\"] = np.mean(e_aucs) if e_aucs else 0.0\n        \n        # Macro AUC (All species)\n        report[\"Macro_AUC\"] = np.mean(list(species_aucs.values()))\n        \n        print(f\"✅ {name} | VeryRare: {report['Very_Rare']:.4f} | Common: {report['Common']:.4f}\")\n        return report\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:32:20.855190Z","iopub.execute_input":"2026-04-28T19:32:20.855512Z","iopub.status.idle":"2026-04-28T19:32:20.871064Z","shell.execute_reply.started":"2026-04-28T19:32:20.855483Z","shell.execute_reply":"2026-04-28T19:32:20.870348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Run Experiments","metadata":{}},{"cell_type":"code","source":"experiments = [\n    (\"Full Pipeline (All On)\", {}),\n    (\"No Logit Adjustment\", {\"use_logit_adj\": False}),\n    (\"No Focal-Soundscape Mixup\", {\"use_mixup\": False}),\n    (\"No Class-Balanced Sampling\", {\"use_sampler\": False}),\n    (\"No Residual Connection\", {\"use_residual\": False}),\n    (\"No Pseudo-Labeling (Focal Only)\", {\"use_pl\": False}),\n    (\"No Temperature Calibration\", {\"use_calib\": False}),\n    (\"Baseline (All Off)\", {\"use_logit_adj\": False, \"use_mixup\": False, \"use_sampler\": False, \"use_residual\": False, \"use_pl\": False, \"use_calib\": False}),\n]\n\nresults = []\nfor name, params in experiments:\n    report = run_experiment(name, **params)\n    results.append(report)\n\nsummary_df = pd.DataFrame(results)\n\n# Calculate Deltas specifically for the \"Very Rare\" bucket\n# This is where we see the REAL impact of our Day 5 upgrades\nsummary_df['VeryRare_Delta'] = summary_df['Very_Rare'] - summary_df.loc[0, 'Very_Rare']\n\n# Rearrange columns for readability\ncols = ['Experiment', 'Macro_AUC', 'Very_Rare', 'Rare', 'Medium', 'Endemic_AUC', 'VeryRare_Delta']\ndisplay(summary_df[cols])\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T19:36:17.574428Z","iopub.execute_input":"2026-04-28T19:36:17.574785Z","iopub.status.idle":"2026-04-28T19:36:30.505225Z","shell.execute_reply.started":"2026-04-28T19:36:17.574755Z","shell.execute_reply":"2026-04-28T19:36:30.503985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}