{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Si une lib manque (Internet doit être ON dans les settings)\n# !pip install timm -q\n\nimport os, gc, ast, random, warnings, time\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nfrom torchaudio.transforms import FrequencyMasking, TimeMasking\nimport timm\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings(\"ignore\")\nprint(\"timm:\", timm.__version__, \"| torch:\", torch.__version__)\nprint(\"GPU dispo:\", torch.cuda.is_available(), \"|\", torch.cuda.get_device_name(0) if torch.cuda.is_available() else \"CPU\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:16.181854Z","iopub.execute_input":"2026-06-25T17:45:16.182144Z","iopub.status.idle":"2026-06-25T17:45:32.384125Z","shell.execute_reply.started":"2026-06-25T17:45:16.182120Z","shell.execute_reply":"2026-06-25T17:45:32.383205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule2**","metadata":{}},{"cell_type":"code","source":"class CFG:\n    # --- Reproductibilité ---\n    seed = 42\n\n    # --- Chemins ---\n    base = \"/kaggle/input/competitions/birdclef-2024\"\n    train_audio = f\"{base}/train_audio\"\n    train_csv   = f\"{base}/train_metadata.csv\"\n    sub_csv     = f\"{base}/sample_submission.csv\"\n\n    # --- Audio ---\n    sample_rate = 32000          # standard BirdCLEF\n    duration    = 5              # secondes (= fenêtre d'évaluation)\n    n_samples   = sample_rate * duration\n\n    # --- Mel-spectrogramme ---\n    n_fft       = 2048\n    hop_length  = 512\n    n_mels      = 128\n    fmin, fmax  = 20, 16000\n    # Largeur temporelle du spectro ≈ n_samples / hop_length + 1 ≈ 313\n\n    # --- Modèle (on changera juste cette ligne pour les 5 modèles) ---\n    model_name  = \"efficientnet_b0\"\n    pretrained  = True\n    in_chans    = 1              # le mel a 1 canal\n    drop_rate      = 0.3         # dropout (anti-overfitting)\n    drop_path_rate = 0.2         # stochastic depth (anti-overfitting)\n\n    # --- Entraînement ---\n    epochs      = 12\n    batch_size  = 64\n    lr          = 1e-3\n    weight_decay= 1e-5\n    n_folds     = 5\n    train_fold  = 0              # on entraîne le fold 0 (boucle possible sur tous)\n    num_workers = 4\n\n    # --- Précalcul des spectrogrammes ---\n    mel_dir           = \"/kaggle/working/melspec\"\n    precompute_seconds = 10   # on garde les 10 premières s de chaque fichier\n    spec_width        = 313   # = n_samples / hop_length + 1 (largeur d'un crop 5 s)\n\n    # --- Anti-overfitting ---\n    label_smoothing = 0.01\n    mixup_prob = 0.5\n    mixup_alpha = 0.5\n    use_weighted_sampler = True\n    early_stop_patience = 4\n\n    # --- Pratique ---\n    DEBUG = False                 \n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\ndef set_seed(seed=42):\n    random.seed(seed); np.random.seed(seed)\n    torch.manual_seed(seed); torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.benchmark = True\nset_seed(CFG.seed)\n\nprint(os.listdir(CFG.base))   # vérifie que les fichiers sont là","metadata":{"execution":{"iopub.status.busy":"2026-06-25T17:45:32.385513Z","iopub.execute_input":"2026-06-25T17:45:32.386438Z","iopub.status.idle":"2026-06-25T17:45:32.405232Z","shell.execute_reply.started":"2026-06-25T17:45:32.386402Z","shell.execute_reply":"2026-06-25T17:45:32.404341Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule3** ","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(CFG.train_csv)\n\n# Liste canonique des classes (ordre = colonnes de la soumission)\nclasses = pd.read_csv(CFG.sub_csv).columns[1:].tolist()   # on enlève 'row_id'\nCFG.n_classes = len(classes)\nlabel2idx = {c: i for i, c in enumerate(classes)}\n\n# Chemin complet de chaque fichier audio\ndf[\"filepath\"] = CFG.train_audio + \"/\" + df[\"filename\"]\n\nprint(\"Nb enregistrements :\", len(df))\nprint(\"Nb classes         :\", CFG.n_classes)\ndf[[\"primary_label\",\"secondary_labels\",\"rating\",\"filename\"]].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:32.406412Z","iopub.execute_input":"2026-06-25T17:45:32.406776Z","iopub.status.idle":"2026-06-25T17:45:32.640599Z","shell.execute_reply.started":"2026-06-25T17:45:32.406716Z","shell.execute_reply":"2026-06-25T17:45:32.639605Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***Cellule4***","metadata":{}},{"cell_type":"code","source":"counts = df[\"primary_label\"].value_counts()\n\nprint(f\"Espèces : {counts.shape[0]}\")\nprint(f\"Max d'enregistrements pour une espèce : {counts.max()}\")\nprint(f\"Min : {counts.min()}\")\nprint(f\"Médiane : {int(counts.median())}\")\nprint(f\"Espèces avec <10 enregistrements : {(counts < 10).sum()}\")\nprint(f\"Espèces avec <5  enregistrements : {(counts < 5).sum()}\")\n\n# Ratio de déséquilibre\nprint(f\"Ratio max/min : {counts.max() / counts.min():.0f}x\")\n\nfig, ax = plt.subplots(1, 2, figsize=(14, 4))\nax[0].plot(counts.values)\nax[0].set_title(\"Distribution triée du nb d'enregistrements / espèce\")\nax[0].set_xlabel(\"Espèce (triée)\"); ax[0].set_ylabel(\"Nb enregistrements\")\nax[1].hist(counts.values, bins=40)\nax[1].set_title(\"Histogramme\")\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:32.642647Z","iopub.execute_input":"2026-06-25T17:45:32.643402Z","iopub.status.idle":"2026-06-25T17:45:33.080334Z","shell.execute_reply.started":"2026-06-25T17:45:32.643375Z","shell.execute_reply":"2026-06-25T17:45:33.079257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule5**","metadata":{}},{"cell_type":"code","source":"df[\"fold\"] = -1\nskf = StratifiedKFold(n_splits=CFG.n_folds, shuffle=True, random_state=CFG.seed)\nfor f, (_, val_idx) in enumerate(skf.split(df, df[\"primary_label\"])):\n    df.loc[val_idx, \"fold\"] = f\n\n# Mode DEBUG : sous-échantillon pour tout tester vite\nif CFG.DEBUG:\n    df = df.groupby(\"primary_label\").head(8).reset_index(drop=True)\n    CFG.epochs = 2\n    print(\">>> MODE DEBUG : échantillon réduit, 2 epochs\")\n\nprint(df[\"fold\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:33.081490Z","iopub.execute_input":"2026-06-25T17:45:33.081886Z","iopub.status.idle":"2026-06-25T17:45:33.121570Z","shell.execute_reply.started":"2026-06-25T17:45:33.081848Z","shell.execute_reply":"2026-06-25T17:45:33.120759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **cellule6**","metadata":{}},{"cell_type":"code","source":"def load_audio(path, sr=CFG.sample_rate):\n    y, _ = librosa.load(path, sr=sr)\n    return y\n\ndef crop_or_pad(y, length, train=True):\n    \"\"\"Découpe un clip de 5 s ; si trop court, on répète le signal (boucle le chant).\"\"\"\n    if len(y) < length:\n        n_repeat = length // len(y) + 1\n        y = np.tile(y, n_repeat)[:length]\n    elif len(y) > length:\n        start = np.random.randint(0, len(y) - length) if train else 0\n        y = y[start:start + length]\n    return y\n\ndef compute_melspec(y):\n    mel = librosa.feature.melspectrogram(\n        y=y, sr=CFG.sample_rate, n_fft=CFG.n_fft,\n        hop_length=CFG.hop_length, n_mels=CFG.n_mels,\n        fmin=CFG.fmin, fmax=CFG.fmax,\n    )\n    mel = librosa.power_to_db(mel, ref=np.max)        # échelle dB (perçue par l'oreille)\n    mel = (mel - mel.min()) / (mel.max() - mel.min() + 1e-6)  # normalisation [0,1]\n    return mel.astype(np.float32)\n\n# Visualisation de contrôle\ny_demo = crop_or_pad(load_audio(df.iloc[0].filepath), CFG.n_samples)\nmel_demo = compute_melspec(y_demo)\nplt.figure(figsize=(8,3))\nplt.imshow(mel_demo, origin=\"lower\", aspect=\"auto\", cmap=\"magma\")\nplt.title(f\"Mel-spectrogramme — {df.iloc[0].primary_label}  shape={mel_demo.shape}\")\nplt.colorbar(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:33.122595Z","iopub.execute_input":"2026-06-25T17:45:33.122975Z","iopub.status.idle":"2026-06-25T17:45:59.031115Z","shell.execute_reply.started":"2026-06-25T17:45:33.122933Z","shell.execute_reply":"2026-06-25T17:45:59.030149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule7**","metadata":{}},{"cell_type":"code","source":"# --- Augmentations waveform (appliquées sur l'audio) ---\ndef add_gaussian_noise(y, min_snr=5, max_snr=20):\n    snr = np.random.uniform(min_snr, max_snr)\n    p = np.mean(y**2)\n    noise = np.random.normal(0, np.sqrt(p / (10**(snr/10))), len(y))\n    return y + noise\n\ndef random_gain(y, low=0.8, high=1.2):\n    return y * np.random.uniform(low, high)\n\ndef time_shift(y, max_frac=0.2):\n    return np.roll(y, int(np.random.uniform(-max_frac, max_frac) * len(y)))\n\ndef augment_waveform(y):\n    if np.random.rand() < 0.5: y = add_gaussian_noise(y)\n    if np.random.rand() < 0.5: y = random_gain(y)\n    if np.random.rand() < 0.5: y = time_shift(y)\n    return y\n\n# --- SpecAugment (masquage fréquence + temps sur le spectro) ---\nfreq_mask = FrequencyMasking(freq_mask_param=24)\ntime_mask = TimeMasking(time_mask_param=40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:59.032238Z","iopub.execute_input":"2026-06-25T17:45:59.032450Z","iopub.status.idle":"2026-06-25T17:45:59.039610Z","shell.execute_reply.started":"2026-06-25T17:45:59.032430Z","shell.execute_reply":"2026-06-25T17:45:59.038939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule8**","metadata":{}},{"cell_type":"code","source":"def parse_secondary(s):\n    try: return ast.literal_eval(s)\n    except: return []\n\nclass BirdDataset(Dataset):\n    def __init__(self, df, train=True):\n        self.df = df.reset_index(drop=True); self.train = train\n    def __len__(self): return len(self.df)\n\n    def _make_target(self, row):\n        t = np.zeros(CFG.n_classes, dtype=np.float32)\n        t[label2idx[row.primary_label]] = 1.0\n        for s in parse_secondary(row.secondary_labels):\n            if s in label2idx: t[label2idx[s]] = 1.0\n        return t * (1 - CFG.label_smoothing) + CFG.label_smoothing / CFG.n_classes\n\n    def __getitem__(self, i):\n        row = self.df.iloc[i]\n        mel = np.load(row.melpath).astype(np.float32)      # (128, T)\n        T, need = mel.shape[1], CFG.spec_width\n        if T < need:                                       # trop court -> on répète\n            mel = np.tile(mel, (1, need // T + 1))[:, :need]\n        elif T > need:                                     # crop temporel\n            start = np.random.randint(0, T - need) if self.train else 0\n            mel = mel[:, start:start + need]\n        mel = torch.tensor(mel).unsqueeze(0)               # (1, 128, 313)\n        if self.train:\n            mel = freq_mask(mel); mel = time_mask(mel)      # SpecAugment\n        return mel, torch.tensor(self._make_target(row))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:59.040471Z","iopub.execute_input":"2026-06-25T17:45:59.040674Z","iopub.status.idle":"2026-06-25T17:45:59.102845Z","shell.execute_reply.started":"2026-06-25T17:45:59.040656Z","shell.execute_reply":"2026-06-25T17:45:59.101856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom joblib import Parallel, delayed\n\nos.makedirs(CFG.mel_dir, exist_ok=True)\n\ndef _mel_path(filename):\n    return os.path.join(CFG.mel_dir, filename.replace(\"/\", \"__\").replace(\".ogg\", \".npy\"))\n\ndef precompute_one(filepath, filename):\n    out = _mel_path(filename)\n    if os.path.exists(out):\n        return\n    try:\n        y, _ = librosa.load(filepath, sr=CFG.sample_rate, duration=CFG.precompute_seconds)\n        mel = librosa.feature.melspectrogram(\n            y=y, sr=CFG.sample_rate, n_fft=CFG.n_fft, hop_length=CFG.hop_length,\n            n_mels=CFG.n_mels, fmin=CFG.fmin, fmax=CFG.fmax)\n        mel = librosa.power_to_db(mel, ref=np.max)\n        mel = (mel - mel.min()) / (mel.max() - mel.min() + 1e-6)\n        np.save(out, mel.astype(np.float16))     # float16 -> 2x moins de stockage\n    except Exception as e:\n        print(\"ERREUR\", filepath, e)\n\nt0 = time.time()\nParallel(n_jobs=4)(delayed(precompute_one)(fp, fn)\n                   for fp, fn in zip(df.filepath, df.filename))\ndf[\"melpath\"] = df[\"filename\"].map(_mel_path)\nprint(f\"Précalcul terminé en {(time.time()-t0)/60:.1f} min | fichiers : {len(os.listdir(CFG.mel_dir))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:45:59.104084Z","iopub.execute_input":"2026-06-25T17:45:59.104403Z","iopub.status.idle":"2026-06-25T17:52:47.033353Z","shell.execute_reply.started":"2026-06-25T17:45:59.104373Z","shell.execute_reply":"2026-06-25T17:52:47.032170Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **   Cellule9**","metadata":{}},{"cell_type":"code","source":"def make_loaders(fold):\n    tr = df[df.fold != fold].reset_index(drop=True)\n    va = df[df.fold == fold].reset_index(drop=True)\n\n    train_ds = BirdDataset(tr, train=True)\n    valid_ds = BirdDataset(va, train=False)\n\n    if CFG.use_weighted_sampler:\n        freq = tr[\"primary_label\"].map(tr[\"primary_label\"].value_counts())\n        weights = 1.0 / freq.values                    # inverse de la fréquence\n        sampler = WeightedRandomSampler(weights, len(weights), replacement=True)\n        train_loader = DataLoader(train_ds, batch_size=CFG.batch_size, sampler=sampler,\n                                  num_workers=CFG.num_workers, pin_memory=True, drop_last=True)\n    else:\n        train_loader = DataLoader(train_ds, batch_size=CFG.batch_size, shuffle=True,\n                                  num_workers=CFG.num_workers, pin_memory=True, drop_last=True)\n\n    valid_loader = DataLoader(valid_ds, batch_size=CFG.batch_size, shuffle=False,\n                              num_workers=CFG.num_workers, pin_memory=True)\n    return train_loader, valid_loader","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:47.036131Z","iopub.execute_input":"2026-06-25T17:52:47.036860Z","iopub.status.idle":"2026-06-25T17:52:47.044059Z","shell.execute_reply.started":"2026-06-25T17:52:47.036818Z","shell.execute_reply":"2026-06-25T17:52:47.042820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule10**","metadata":{}},{"cell_type":"code","source":"import torch, torch.nn as nn, torch.nn.functional as F, timm, gc\n\n# ───────────────── Helper : création tolérante aux kwargs ──────────────────\ndef _create_backbone(name, **kwargs):\n    \"\"\"Essaie de créer le backbone avec tous les kwargs ;\n    si drop_path_rate est refusé, réessaie sans lui.\"\"\"\n    try:\n        return timm.create_model(name, **kwargs)\n    except TypeError as e:\n        if \"drop_path_rate\" in str(e):\n            kwargs.pop(\"drop_path_rate\", None)\n            return timm.create_model(name, **kwargs)\n        raise  # autre erreur → on la remonte\n\n\n# ───────────────────────── 1) Modèle CNN générique ─────────────────────────\nclass BirdModel(nn.Module):\n    def __init__(self, name, pretrained=CFG.pretrained):\n        super().__init__()\n        self.backbone = _create_backbone(\n            name, pretrained=pretrained, in_chans=CFG.in_chans,\n            num_classes=0, global_pool=\"avg\",\n            drop_rate=CFG.drop_rate, drop_path_rate=CFG.drop_path_rate,\n        )\n        self.head = nn.Linear(self.backbone.num_features, CFG.n_classes)\n\n    def forward(self, x):\n        return self.head(self.backbone(x))\n\n\n# ───────────────────────── 2) Modèle SED (attention temporelle) ─────────────\nclass AttHead(nn.Module):\n    def __init__(self, in_chans, n_classes, p=0.3):\n        super().__init__()\n        self.dropout = nn.Dropout(p)\n        self.att = nn.Conv1d(in_chans, n_classes, 1)\n        self.cla = nn.Conv1d(in_chans, n_classes, 1)\n\n    def forward(self, feat):\n        feat = self.dropout(feat)\n        norm_att = torch.softmax(torch.tanh(self.att(feat)), dim=-1)\n        frame_logits = self.cla(feat)\n        return torch.sum(norm_att * frame_logits, dim=-1)\n\n\nclass BirdSED(nn.Module):\n    def __init__(self, name=\"efficientnet_b0\", pretrained=CFG.pretrained):\n        super().__init__()\n        self.backbone = _create_backbone(\n            name, pretrained=pretrained, in_chans=CFG.in_chans,\n            num_classes=0, global_pool=\"\",\n            drop_rate=CFG.drop_rate, drop_path_rate=CFG.drop_path_rate,\n        )\n        n_feat = self.backbone.num_features\n        self.bn = nn.BatchNorm1d(n_feat)\n        self.fc = nn.Linear(n_feat, n_feat)\n        self.head = AttHead(n_feat, CFG.n_classes, p=CFG.drop_rate)\n\n    def forward(self, x):\n        feat = self.backbone.forward_features(x)\n        feat = feat.mean(dim=2)\n        feat = self.bn(feat)\n        feat = F.relu(self.fc(feat.transpose(1, 2))).transpose(1, 2)\n        return self.head(feat)\n\n# ───────────────────────── 3) Sélecteur de modèle ──────────────────────────\ndef build_model(name, architecture=\"cnn\", pretrained=None):\n    p = CFG.pretrained if pretrained is None else pretrained\n    return BirdSED(name, p) if architecture == \"sed\" else BirdModel(name, p)\n\n\n# ───────────────────────── 4) Le zoo (4 CNN + 1 SED) ───────────────────────\nMODELS = {\n    \"efficientnet_b0\":     (\"cnn\", \"CNN MBConv — référence BirdCLEF\"),\n    \"convnext_nano\":       (\"cnn\", \"CNN moderne — gros noyaux + LayerNorm\"),\n    \"seresnext26t_32x4d\":  (\"cnn\", \"CNN + attention de canaux (SE)\"),\n    \"densenet121\":         (\"cnn\", \"CNN connectivité dense — économe en données\"),\n    \"efficientnet_b0\":     (\"cnn\", \"...\"),  # (clé dupliquée volontairement écrasée, voir note)\n}\n# NB : un dict ne peut pas avoir 2 fois la même clé. Le SED réutilise efficientnet_b0\n# comme backbone mais via architecture='sed', donc on le déclare à part :\nMODELS = {\n    \"efficientnet_b0\":     (\"cnn\", \"CNN MBConv — référence BirdCLEF\"),\n    \"convnext_nano\":       (\"cnn\", \"CNN moderne — gros noyaux + LayerNorm\"),\n    \"seresnext26t_32x4d\":  (\"cnn\", \"CNN + attention de canaux (SE)\"),\n    \"densenet121\":         (\"cnn\", \"CNN connectivité dense — économe en données\"),\n    \"sed_efficientnet_b0\": (\"sed\", \"SED — attention temporelle (top teams)\"),\n}\n\n# ───────────────────────── 5) Test des dimensions (sans télécharger) ────────\ndummy = torch.randn(2, 1, CFG.n_mels, 313).to(device)\nfor key, (arch, why) in MODELS.items():\n    backbone = \"efficientnet_b0\" if key.startswith(\"sed_\") else key\n    m = build_model(backbone, architecture=arch, pretrained=False).to(device)\n    out = m(dummy)\n    assert out.shape == (2, CFG.n_classes), f\"{key} -> {out.shape}\"\n    print(f\" {key:22s} {tuple(out.shape)}  | {why}\")\n    del m; gc.collect(); torch.cuda.empty_cache()\n\nprint(\"\\nTous les modèles renvoient bien (B, 182) — prêts pour l'entraînement.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:47.045179Z","iopub.execute_input":"2026-06-25T17:52:47.045560Z","iopub.status.idle":"2026-06-25T17:52:57.731191Z","shell.execute_reply.started":"2026-06-25T17:52:47.045511Z","shell.execute_reply":"2026-06-25T17:52:57.730397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule11**","metadata":{}},{"cell_type":"code","source":"criterion = nn.BCEWithLogitsLoss()   # multi-label (primary + secondary)\n\ndef mixup(x, y, alpha=CFG.mixup_alpha):\n    lam = np.random.beta(alpha, alpha)\n    idx = torch.randperm(x.size(0), device=x.device)\n    return lam*x + (1-lam)*x[idx], lam*y + (1-lam)*y[idx]\n\n# Trouve cette fonction et ajoute y_true = y_true.astype(int)\ndef macro_auc(y_true, y_pred):\n    y_true = (y_true >= 0.5).astype(int)   # ← seuil 0.5 au lieu de .astype(int)\n    aucs = []\n    for c in range(y_true.shape[1]):\n        if y_true[:, c].sum() > 0:\n            aucs.append(roc_auc_score(y_true[:, c], y_pred[:, c]))\n    return float(np.mean(aucs)) if aucs else 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:57.732695Z","iopub.execute_input":"2026-06-25T17:52:57.733012Z","iopub.status.idle":"2026-06-25T17:52:57.739788Z","shell.execute_reply.started":"2026-06-25T17:52:57.732987Z","shell.execute_reply":"2026-06-25T17:52:57.738782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule12**","metadata":{}},{"cell_type":"code","source":"def train_one_epoch(model, loader, optimizer, scheduler, scaler):\n    model.train(); running = 0.0\n    for x, y in loader:\n        x, y = x.to(device), y.to(device)\n        if np.random.rand() < CFG.mixup_prob:\n            x, y = mixup(x, y)\n        optimizer.zero_grad()\n        with torch.cuda.amp.autocast():\n            loss = criterion(model(x), y)\n        scaler.scale(loss).backward()\n        scaler.unscale_(optimizer)\n        torch.nn.utils.clip_grad_norm_(model.parameters(), 5.0)\n        scaler.step(optimizer); scaler.update(); scheduler.step()\n        running += loss.item()\n    return running / len(loader)\n\n@torch.no_grad()\ndef validate(model, loader):\n    model.eval(); preds, trues, running = [], [], 0.0\n    for x, y in loader:\n        x = x.to(device)\n        with torch.cuda.amp.autocast():\n            out = model(x)\n            running += criterion(out, y.to(device)).item()\n        preds.append(out.sigmoid().cpu().numpy()); trues.append(y.numpy())\n    preds, trues = np.concatenate(preds), np.concatenate(trues)\n    return running / len(loader), macro_auc(trues, preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:57.741043Z","iopub.execute_input":"2026-06-25T17:52:57.741437Z","iopub.status.idle":"2026-06-25T17:52:57.770625Z","shell.execute_reply.started":"2026-06-25T17:52:57.741413Z","shell.execute_reply":"2026-06-25T17:52:57.769653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader, valid_loader = make_loaders(CFG.train_fold)\n\n# Inspecte un batch de validation\nx, y = next(iter(valid_loader))\nprint(\"y shape   :\", y.shape)\nprint(\"y dtype   :\", y.dtype)\nprint(\"y min/max :\", y.min().item(), y.max().item())\nprint(\"positifs  :\", (y > 0).sum().item(), \"sur\", y.numel())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:57.771790Z","iopub.execute_input":"2026-06-25T17:52:57.772197Z","iopub.status.idle":"2026-06-25T17:52:58.377606Z","shell.execute_reply.started":"2026-06-25T17:52:57.772171Z","shell.execute_reply":"2026-06-25T17:52:58.376485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Cellule13**","metadata":{}},{"cell_type":"code","source":"def run(model_key, fold=CFG.train_fold):\n    arch, _ = MODELS[model_key]\n    backbone = \"efficientnet_b0\" if model_key.startswith(\"sed_\") else model_key\n    print(f\"\\n===== {model_key} ({arch}) | fold {fold} =====\")\n    set_seed(CFG.seed)\n    train_loader, valid_loader = make_loaders(fold)\n    model = build_model(backbone, architecture=arch).to(device)\n\n    optimizer = torch.optim.AdamW(model.parameters(), lr=CFG.lr, weight_decay=CFG.weight_decay)\n    scheduler = torch.optim.lr_scheduler.OneCycleLR(\n        optimizer, max_lr=CFG.lr, epochs=CFG.epochs, steps_per_epoch=len(train_loader))\n    scaler = torch.cuda.amp.GradScaler()\n\n    history = {\"train_loss\": [], \"val_loss\": [], \"val_auc\": []}\n    best_auc, patience = 0.0, 0\n    for epoch in range(CFG.epochs):\n        t0 = time.time()\n        tr = train_one_epoch(model, train_loader, optimizer, scheduler, scaler)\n        vl, va = validate(model, valid_loader)\n        history[\"train_loss\"].append(tr); history[\"val_loss\"].append(vl); history[\"val_auc\"].append(va)\n        print(f\"Ep {epoch+1:02d} | train {tr:.4f} | val {vl:.4f} | AUC {va:.4f} | {time.time()-t0:.0f}s\")\n        if va > best_auc:\n            best_auc, patience = va, 0\n            torch.save(model.state_dict(), f\"{model_key}_fold{fold}.pth\")\n        else:\n            patience += 1\n            if patience >= CFG.early_stop_patience:\n                print(\"    Early stopping\"); break\n    print(f\"Meilleur AUC {model_key} : {best_auc:.4f}\")\n    del model; gc.collect(); torch.cuda.empty_cache()\n    return best_auc, history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:58.379404Z","iopub.execute_input":"2026-06-25T17:52:58.379801Z","iopub.status.idle":"2026-06-25T17:52:58.389809Z","shell.execute_reply.started":"2026-06-25T17:52:58.379704Z","shell.execute_reply":"2026-06-25T17:52:58.388668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(model_key, h):\n    ep = range(1, len(h[\"train_loss\"]) + 1)\n    fig, ax = plt.subplots(1, 2, figsize=(12, 4))\n    ax[0].plot(ep, h[\"train_loss\"], \"o-\", label=\"train\")\n    ax[0].plot(ep, h[\"val_loss\"],   \"o-\", label=\"val\")\n    ax[0].set_title(f\"{model_key} — Loss\"); ax[0].set_xlabel(\"epoch\"); ax[0].legend(); ax[0].grid(alpha=.3)\n    ax[1].plot(ep, h[\"val_auc\"], \"o-\", color=\"green\")\n    ax[1].set_title(f\"{model_key} — Val macro-AUC\"); ax[1].set_xlabel(\"epoch\"); ax[1].grid(alpha=.3)\n    plt.tight_layout(); plt.show()\n\nhistories, scores = {}, {}\nfor k in MODELS:\n    best, h = run(k)\n    scores[k], histories[k] = best, h\n    plot_history(k, h)\n\n# --- Comparaison finale des 5 modèles ---\nplt.figure(figsize=(9, 4))\nbars = plt.bar(scores.keys(), scores.values(), color=\"steelblue\")\nplt.bar_label(bars, fmt=\"%.3f\"); plt.ylabel(\"Best Val macro-AUC\")\nplt.xticks(rotation=25, ha=\"right\"); plt.title(\"Comparaison des 5 modèles\")\nplt.tight_layout(); plt.show()\n\nprint(\"Scores :\", {k: round(v, 4) for k, v in scores.items()})\nprint(\"Ensemble (moyenne) attendu > meilleur modèle solo\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T17:52:58.391023Z","iopub.execute_input":"2026-06-25T17:52:58.391427Z","iopub.status.idle":"2026-06-25T18:55:55.549034Z","shell.execute_reply.started":"2026-06-25T17:52:58.391402Z","shell.execute_reply":"2026-06-25T18:55:55.547907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping code espèce -> nom commun + nom scientifique\nmeta = pd.read_csv(CFG.train_csv)\ntry:\n    code2name = meta.groupby(\"primary_label\")[\"common_name\"].first().to_dict()\n    code2sci  = meta.groupby(\"primary_label\")[\"scientific_name\"].first().to_dict()\nexcept KeyError:\n    code2name = {c: c for c in classes}; code2sci = {c: \"\" for c in classes}\n\n# Charge les 5 modèles entraînés (depuis les .pth sauvegardés)\ndef load_trained(model_key, fold=0):\n    arch, _ = MODELS[model_key]\n    backbone = \"efficientnet_b0\" if model_key.startswith(\"sed_\") else model_key\n    m = build_model(backbone, architecture=arch, pretrained=False).to(device)\n    m.load_state_dict(torch.load(f\"{model_key}_fold{fold}.pth\", map_location=device))\n    m.eval()\n    return m\n\nENSEMBLE = []\nfor k in MODELS:\n    try:\n        ENSEMBLE.append(load_trained(k)); print(f\" chargé : {k}\")\n    except Exception as e:\n        print(f\" ignoré {k} : {e}\")\nprint(f\"\\nEnsemble prêt : {len(ENSEMBLE)} modèles\")\n\n@torch.no_grad()\ndef predict_audio(path, topk=5, max_chunks=6):\n    \"\"\"Découpe en fenêtres 5 s, moyenne l'ensemble sur chaque fenêtre.\"\"\"\n    y, _ = librosa.load(path, sr=CFG.sample_rate)\n    seg = CFG.n_samples\n    if len(y) < seg:\n        y = np.tile(y, seg // len(y) + 1)[:seg]\n    starts = list(range(0, len(y) - seg + 1, seg))[:max_chunks] or [0]\n    all_probs = []\n    for s in starts:\n        c = y[s:s + seg]\n        if len(c) < seg: c = np.pad(c, (0, seg - len(c)))\n        x = torch.tensor(compute_melspec(c)).unsqueeze(0).unsqueeze(0).to(device)  # (1,1,128,313)\n        p = np.mean([torch.sigmoid(m(x)).cpu().numpy()[0] for m in ENSEMBLE], axis=0)\n        all_probs.append(p)\n    probs = np.mean(all_probs, axis=0)\n    top = probs.argsort()[::-1][:topk]\n    return probs, [(classes[i], float(probs[i])) for i in top]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T18:55:55.550480Z","iopub.execute_input":"2026-06-25T18:55:55.550891Z","iopub.status.idle":"2026-06-25T18:55:57.161243Z","shell.execute_reply.started":"2026-06-25T18:55:55.550840Z","shell.execute_reply":"2026-06-25T18:55:57.160215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install gradio -q\nimport gradio as gr\n\ndef interface_fn(audio_path):\n    if audio_path is None:\n        return {}, None, \"Charge un fichier audio (.ogg / .wav / .mp3) puis clique Identifier.\"\n    probs, top = predict_audio(audio_path)\n\n    # 1) Dictionnaire {nom commun: confiance} pour le composant Label\n    labels = {f\"{code2name.get(c, c)}\": p for c, p in top}\n\n    # 2) Spectrogramme de la 1re fenêtre\n    y, _ = librosa.load(audio_path, sr=CFG.sample_rate)\n    y = crop_or_pad(y, CFG.n_samples, train=False)\n    fig, ax = plt.subplots(figsize=(7, 3))\n    ax.imshow(compute_melspec(y), origin=\"lower\", aspect=\"auto\", cmap=\"magma\")\n    ax.set_title(\"Mel-spectrogramme analysé\"); ax.set_xlabel(\"Temps\"); ax.set_ylabel(\"Fréquence (mel)\")\n    plt.tight_layout()\n\n    # 3) Détail texte avec noms scientifiques\n    detail = \"### Top-5 espèces détectées\\n\\n| Espèce | Nom scientifique | Confiance |\\n|---|---|---|\\n\"\n    detail += \"\\n\".join(f\"| {code2name.get(c,c)} | *{code2sci.get(c,'')}* | {p*100:.1f}% |\" for c, p in top)\n    return labels, fig, detail\n\nwith gr.Blocks(theme=gr.themes.Soft(primary_hue=\"emerald\"), title=\"BirdCLEF — Identification de cris d'oiseaux\") as demo:\n    gr.Markdown(\n        \"#  Identification d'oiseaux par le chant\\n\"\n        \"Surveillance acoustique de la biodiversité (Ghats occidentaux, 182 espèces). \"\n        \"Charge un enregistrement, l'**ensemble de 5 modèles** prédit les espèces présentes.\"\n    )\n    with gr.Row():\n        with gr.Column():\n            audio_in = gr.Audio(type=\"filepath\", label=\" Enregistrement audio\")\n            btn = gr.Button(\" Identifier l'espèce\", variant=\"primary\")\n        with gr.Column():\n            label_out = gr.Label(num_top_classes=5, label=\"Espèces probables\")\n    spec_out = gr.Plot(label=\"Spectrogramme\")\n    detail_out = gr.Markdown()\n\n    btn.click(interface_fn, inputs=audio_in, outputs=[label_out, spec_out, detail_out])\n\n    # Exemples cliquables (3 espèces du dataset) pour tester sans uploader\n    ex = meta.groupby(\"primary_label\").first().reset_index().sample(3, random_state=0)\n    examples = [[p] for p in (CFG.train_audio + \"/\" + ex[\"filename\"]).tolist()]\n    gr.Examples(examples, inputs=audio_in, label=\"Exemples à tester\")\n\ndemo.launch(share=True)   # 'share=True' -> lien public partageable","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-25T18:55:57.162571Z","iopub.execute_input":"2026-06-25T18:55:57.163203Z","iopub.status.idle":"2026-06-25T18:56:16.169302Z","shell.execute_reply.started":"2026-06-25T18:55:57.163175Z","shell.execute_reply":"2026-06-25T18:56:16.168478Z"}},"outputs":[],"execution_count":null}]}