{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":12073001,"sourceType":"datasetVersion","datasetId":7599713},{"sourceId":243934463,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:29:47.792456Z","iopub.execute_input":"2025-06-05T17:29:47.792907Z","iopub.status.idle":"2025-06-05T17:30:16.957103Z","shell.execute_reply.started":"2025-06-05T17:29:47.792883Z","shell.execute_reply":"2025-06-05T17:30:16.956124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport librosa\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport timm\n\n# Config\nclass CFG:\n    sample_rate = 32000\n    duration = 5  # seconds\n    n_mels = 128\n    n_fft = 2048\n    hop_length = 512\n    num_classes = 206\n    model_name = 'efficientnet_b0'\n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    test_dir = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:30:16.958710Z","iopub.execute_input":"2025-06-05T17:30:16.958983Z","iopub.status.idle":"2025-06-05T17:30:16.966025Z","shell.execute_reply.started":"2025-06-05T17:30:16.958963Z","shell.execute_reply":"2025-06-05T17:30:16.964833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EfficientNetFrozen(nn.Module):\n    def __init__(self, model_name='efficientnet_b0', n_classes=206):\n        super().__init__()\n        self.backbone = timm.create_model(model_name, pretrained=False, in_chans=1, num_classes=0)\n        self.classifier = nn.Linear(self.backbone.num_features, n_classes)\n\n    def forward(self, x):\n        x = self.backbone(x)\n        x = self.classifier(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:30:16.967319Z","iopub.execute_input":"2025-06-05T17:30:16.967787Z","iopub.status.idle":"2025-06-05T17:30:16.987096Z","shell.execute_reply.started":"2025-06-05T17:30:16.967749Z","shell.execute_reply":"2025-06-05T17:30:16.985706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio_to_logmel(y, cfg):\n    mel = librosa.feature.melspectrogram(\n        y=y,\n        sr=cfg.sample_rate,\n        n_fft=cfg.n_fft,\n        hop_length=cfg.hop_length,\n        n_mels=cfg.n_mels\n    )\n    logmel = librosa.power_to_db(mel)\n    # Normalize per frequency bin\n    logmel = (logmel - logmel.mean(axis=1, keepdims=True)) / (logmel.std(axis=1, keepdims=True) + 1e-6)\n    return logmel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:30:16.988435Z","iopub.execute_input":"2025-06-05T17:30:16.988780Z","iopub.status.idle":"2025-06-05T17:30:17.013467Z","shell.execute_reply.started":"2025-06-05T17:30:16.988757Z","shell.execute_reply":"2025-06-05T17:30:17.012360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport torch\n\ndef predict_on_soundscape_with_smoothing(\n    file_path,\n    model,\n    cfg,\n    smoothing_window: int = 5\n):\n    \"\"\"\n    Predict on one soundscape, then apply a moving‐average filter along the time axis\n    to smooth per-class probabilities across adjacent chunks.\n\n    Args:\n      file_path: Path to the .ogg soundscape.\n      model: Your trained torch.nn.Module.\n      cfg: CFG object (contains sample_rate, duration, etc.).\n      smoothing_window: Number of chunks over which to average. Must be an odd integer.\n                        (e.g. 3 => average over [i-1, i, i+1].)\n    Returns:\n      row_ids: List of strings (e.g. \"soundscape_5\", \"soundscape_10\", …)\n      smoothed_preds: List of 1D NumPy arrays (length=num_classes), one per chunk,\n                      after smoothing.\n    \"\"\"\n    # Load entire audio (mono)\n    y, _ = librosa.load(file_path, sr=cfg.sample_rate)\n    chunk_size = cfg.duration * cfg.sample_rate\n    num_chunks = len(y) // chunk_size\n\n    raw_preds = []   # will store one (num_classes,) array per chunk\n    stem = Path(file_path).stem\n\n    # 1) Collect raw predictions for each chunk\n    for i in range(num_chunks):\n        start = i * chunk_size\n        end = start + chunk_size\n        chunk = y[start:end]\n\n        # If last chunk is shorter (shouldn’t happen if len(y)//chunk_size is exact),\n        # pad with zeros\n        if len(chunk) < chunk_size:\n            chunk = np.pad(chunk, (0, chunk_size - len(chunk)))\n\n        # Convert to log‐mel, then tensor → model → sigmoid\n        logmel = audio_to_logmel(chunk, cfg)\n        tensor = (\n            torch.tensor(logmel)\n            .unsqueeze(0)\n            .unsqueeze(0)\n            .float()\n            .to(cfg.device)\n        )\n\n        with torch.no_grad():\n            output = model(tensor)          # shape: (1, num_classes)\n            prob   = torch.sigmoid(output)  # shape: (1, num_classes)\n            prob   = prob.cpu().numpy().squeeze()  # → (num_classes,)\n        raw_preds.append(prob)\n\n    # If there are no chunks (very short file), return empty\n    if len(raw_preds) == 0:\n        return [], []\n\n    raw_preds = np.stack(raw_preds, axis=0)  \n    # raw_preds shape: (num_chunks, num_classes)\n\n    # 2) Build a uniform (moving‐average) kernel of length smoothing_window\n    if smoothing_window < 1 or smoothing_window % 2 == 0:\n        raise ValueError(\"smoothing_window must be a positive odd integer\")\n    kernel = np.ones(smoothing_window, dtype=np.float32) / smoothing_window\n\n    # 3) Apply 1D convolution (per class) along the “time” axis.\n    #    We pad at both ends with 'edge' (i.e. repeat the first/last) so that the\n    #    smoothed array has the same length = num_chunks.\n    pad_len = smoothing_window // 2\n    smoothed_preds = np.empty_like(raw_preds)\n\n    # For each class idx, convolve its 1D time series with the uniform kernel.\n    for cls in range(raw_preds.shape[1]):\n        series = raw_preds[:, cls]  # shape: (num_chunks,)\n        # pad with edge values so that index 0 uses series[0], index -1 uses series[-1], etc.\n        padded = np.pad(series, (pad_len, pad_len), mode=\"edge\")\n        conved = np.convolve(padded, kernel, mode=\"valid\")  \n        # After 'valid', conved.shape = (num_chunks,)\n        smoothed_preds[:, cls] = conved\n\n    # 4) Build row_ids and turn the smoothed_preds back into a list of arrays\n    row_ids = []\n    smoothed_list = []\n    for i in range(num_chunks):\n        timestamp = (i + 1) * cfg.duration\n        row_id = f\"{stem}_{timestamp}\"\n        row_ids.append(row_id)\n        smoothed_list.append(smoothed_preds[i])\n\n    return row_ids, smoothed_list\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:30:17.015551Z","iopub.execute_input":"2025-06-05T17:30:17.015953Z","iopub.status.idle":"2025-06-05T17:30:17.050894Z","shell.execute_reply.started":"2025-06-05T17:30:17.015919Z","shell.execute_reply":"2025-06-05T17:30:17.049508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_submission_with_smoothing(model, cfg, smoothing_window=5):\n    model.eval()\n    test_files = list(Path(cfg.test_dir).glob(\"*.ogg\"))\n\n    all_row_ids = []\n    all_preds   = []\n\n    for file_path in tqdm(test_files):\n        # Use the \"with_smoothing\" version here:\n        row_ids, preds = predict_on_soundscape_with_smoothing(\n            file_path, model, cfg, smoothing_window=smoothing_window\n        )\n        all_row_ids.extend(row_ids)\n        all_preds.extend(preds)\n\n    # Build DataFrame just like before, but using smoothed preds\n    pred_df = pd.DataFrame(\n        all_preds, \n        columns=[f\"class_{i}\" for i in range(cfg.num_classes)]\n    )\n    pred_df.insert(0, \"row_id\", all_row_ids)\n\n    sample = pd.read_csv(cfg.submission_csv)\n    pred_df = pred_df.set_index(\"row_id\")\n    sample  = sample.set_index(\"row_id\")\n    final   = sample.copy()\n    final.loc[pred_df.index] = pred_df\n    final   = final.reset_index()\n    final.to_csv(\"submission.csv\", index=False)\n    print(\"✅ submission_smoothed.csv created.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:30:17.052094Z","iopub.execute_input":"2025-06-05T17:30:17.052598Z","iopub.status.idle":"2025-06-05T17:30:17.098977Z","shell.execute_reply.started":"2025-06-05T17:30:17.052548Z","shell.execute_reply":"2025-06-05T17:30:17.097339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load model\nmodel = EfficientNetFrozen(model_name=CFG.model_name, n_classes=CFG.num_classes)\nmodel.load_state_dict(torch.load(\"/kaggle/input/efficientnet-v4/efficientnet_b0_frozen_overall_best.pth\", map_location=CFG.device))\nmodel.to(CFG.device)\n\n# Generate predictions\n# Generate submission\ngenerate_submission_with_smoothing(\n    model=model,\n    cfg=CFG,\n    smoothing_window=3\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T17:30:17.100163Z","iopub.execute_input":"2025-06-05T17:30:17.100964Z","iopub.status.idle":"2025-06-05T17:30:17.375872Z","shell.execute_reply.started":"2025-06-05T17:30:17.100929Z","shell.execute_reply":"2025-06-05T17:30:17.374749Z"}},"outputs":[],"execution_count":null}]}