{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport math\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport soundfile as sf\nimport torch\nimport torch.nn as nn\nimport torchvision.models as models\nfrom pathlib import Path\nfrom torch.utils.data import Dataset\nfrom tqdm.notebook import tqdm\nfrom matplotlib import pyplot as plt\n\n# Конфигурация\nN_CLASSES = 397\nSAMPLE_RATE = 32000\nCLIP_LEN = 5\nCONF_THRESH = 0.2  # Порог уверенности\n\n# Определение устройства\nACCELERATOR = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {ACCELERATOR}\")\n\n# Пути к данным\nBASE_DIR = Path(\"../input/birdclef-2021\")\nTEST_DIR = BASE_DIR / \"test_soundscapes\"\nSUB_FILE = BASE_DIR / \"sample_submission.csv\"\nLABELS_FILE = None\n\n# Проверка режима (Submit vs Offline Debug)\nif len(list(TEST_DIR.glob(\"*.ogg\"))) == 0:\n    print(\"Debug mode: using train_soundscapes\")\n    TEST_DIR = BASE_DIR / \"train_soundscapes\"\n    SUB_FILE = None\n    LABELS_FILE = BASE_DIR / \"train_soundscape_labels.csv\"\n\nprint(f\"Reading audio from: {TEST_DIR}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:41:55.285957Z","iopub.execute_input":"2025-12-15T19:41:55.286550Z","iopub.status.idle":"2025-12-15T19:41:59.045792Z","shell.execute_reply.started":"2025-12-15T19:41:55.286527Z","shell.execute_reply":"2025-12-15T19:41:59.044894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/input/resnest50-fast-package/resnest-0.0.6b20200701/resnest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:03.116236Z","iopub.execute_input":"2025-12-15T19:42:03.117173Z","iopub.status.idle":"2025-12-15T19:42:03.250477Z","shell.execute_reply.started":"2025-12-15T19:42:03.117144Z","shell.execute_reply":"2025-12-15T19:42:03.249478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\n# Добавляем пути к локальному пакету resnest\npackage_path = \"/kaggle/input/resnest50-fast-package/resnest-0.0.6b20200701\"\nsys.path.append(package_path)\nsys.path.append(os.path.join(package_path, \"resnest\"))\n\nfrom resnest.torch import resnest50\nprint(\"ResNeSt library loaded.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:04.095877Z","iopub.execute_input":"2025-12-15T19:42:04.096778Z","iopub.status.idle":"2025-12-15T19:42:04.116310Z","shell.execute_reply.started":"2025-12-15T19:42:04.096744Z","shell.execute_reply":"2025-12-15T19:42:04.115769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LOCATION_BIRDS = {}\n\nif LABELS_FILE and LABELS_FILE.exists():\n    df_labels = pd.read_csv(LABELS_FILE)\n    for row in df_labels.itertuples(index=False):\n        if isinstance(row.birds, str) and row.birds != \"nocall\":\n            bird_list = row.birds.split()\n            if row.site not in LOCATION_BIRDS:\n                LOCATION_BIRDS[row.site] = set()\n            LOCATION_BIRDS[row.site].update(bird_list)\n\n    print(\"Unique birds per location:\")\n    for loc, birds in LOCATION_BIRDS.items():\n        print(f\"Location {loc}: {len(birds)}\")\nelse:\n    print(\"Labels file not found, skipping stats.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:04.989548Z","iopub.execute_input":"2025-12-15T19:42:04.990245Z","iopub.status.idle":"2025-12-15T19:42:05.010176Z","shell.execute_reply.started":"2025-12-15T19:42:04.990219Z","shell.execute_reply":"2025-12-15T19:42:05.009420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_list = []\nfor p in TEST_DIR.glob(\"*.ogg\"):\n    parts = p.stem.split(\"_\")\n    file_list.append((p.stem, parts[0], parts[1], parts[2], p))\n\nmeta_df = pd.DataFrame(file_list, columns=[\"filename\", \"id\", \"site\", \"date\", \"filepath\"])\nprint(f\"Total soundscapes: {len(meta_df)}\")\nmeta_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:07.941239Z","iopub.execute_input":"2025-12-15T19:42:07.941792Z","iopub.status.idle":"2025-12-15T19:42:07.958094Z","shell.execute_reply.started":"2025-12-15T19:42:07.941768Z","shell.execute_reply":"2025-12-15T19:42:07.957425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioToSpec:\n    def __init__(self, rate, mels, f_min, f_max, **config):\n        self.rate = rate\n        self.mels = mels\n        self.f_min = f_min\n        self.f_max = f_max\n        config[\"n_fft\"] = config.get(\"n_fft\", rate // 10)\n        config[\"hop_length\"] = config.get(\"hop_length\", rate // 40)\n        self.config = config\n\n    def __call__(self, signal):\n        spec = librosa.feature.melspectrogram(\n            y=signal, sr=self.rate, n_mels=self.mels, \n            fmin=self.f_min, fmax=self.f_max, **self.config\n        )\n        return librosa.power_to_db(spec).astype(np.float32)\n\ndef norm_image(img, eps=1e-6):\n    mean_val = img.mean()\n    std_val = img.std()\n    img = (img - mean_val) / (std_val + eps)\n    \n    min_v, max_v = img.min(), img.max()\n    if (max_v - min_v) > eps:\n        img = np.clip(img, min_v, max_v)\n        img = 255 * (img - min_v) / (max_v - min_v)\n        return img.astype(np.uint8)\n    return np.zeros_like(img, dtype=np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:09.916163Z","iopub.execute_input":"2025-12-15T19:42:09.916431Z","iopub.status.idle":"2025-12-15T19:42:09.923616Z","shell.execute_reply.started":"2025-12-15T19:42:09.916413Z","shell.execute_reply":"2025-12-15T19:42:09.922698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class InferenceDataset(Dataset):\n    def __init__(self, df, sr=SAMPLE_RATE, duration=CLIP_LEN):\n        self.df = df\n        self.sr = sr\n        self.duration = duration\n        self.chunk_len = int(duration * sr)\n        self.converter = AudioToSpec(sr=sr, mels=128, f_min=0, f_max=sr//2)\n\n    def __len__(self):\n        return len(self.df)\n\n    def _process_chunk(self, chunk):\n        spec = self.converter(chunk)\n        img = norm_image(spec)\n        img = img.astype(\"float32\", copy=False) / 255.0\n        return np.stack([img, img, img])\n\n    def __getitem__(self, idx):\n        path = self.df.loc[idx, \"filepath\"]\n        raw_audio, _ = sf.read(path, dtype=\"float32\")\n        \n        # Если частота не совпадает, ресемплим (но обычно в датасете 32k)\n        if len(raw_audio) > 0:\n            # Просто заглушка, librosa.resample тяжелый, надеемся на совпадение SR\n            pass \n\n        # Нарезка\n        chunks = []\n        for i in range(0, len(raw_audio), self.chunk_len):\n            if i + self.chunk_len <= len(raw_audio):\n                chunks.append(raw_audio[i : i + self.chunk_len])\n        \n        # Если файл пустой или странный\n        if not chunks:\n            chunks = [np.zeros(self.chunk_len, dtype=np.float32)]\n\n        images = [self._process_chunk(c) for c in chunks]\n        return np.stack(images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:11.750451Z","iopub.execute_input":"2025-12-15T19:42:11.750748Z","iopub.status.idle":"2025-12-15T19:42:11.758621Z","shell.execute_reply.started":"2025-12-15T19:42:11.750725Z","shell.execute_reply":"2025-12-15T19:42:11.757884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class InferenceDataset(Dataset):\n    def __init__(self, df, sr=SAMPLE_RATE, duration=CLIP_LEN):\n        self.df = df\n        self.sr = sr\n        self.duration = duration\n        self.chunk_len = int(duration * sr)\n        # ИСПРАВЛЕНИЕ: передаем rate=sr (было sr=sr)\n        self.converter = AudioToSpec(rate=sr, mels=128, f_min=0, f_max=sr//2)\n\n    def __len__(self):\n        return len(self.df)\n\n    def _process_chunk(self, chunk):\n        spec = self.converter(chunk)\n        img = norm_image(spec)\n        img = img.astype(\"float32\", copy=False) / 255.0\n        return np.stack([img, img, img])\n\n    def __getitem__(self, idx):\n        path = self.df.loc[idx, \"filepath\"]\n        # Читаем аудио. Если файл битый, вернем заглушку\n        try:\n            raw_audio, _ = sf.read(path, dtype=\"float32\")\n        except:\n            raw_audio = np.array([])\n        \n        chunks = []\n        for i in range(0, len(raw_audio), self.chunk_len):\n            # Берем куски только полной длины (или паддим, если нужно, но здесь просто режем)\n            # В оригинале мы паддили короткие, но для простоты берем полные 5 сек\n            # Но чтобы не терять хвосты, лучше дополнить нулями, если кусок < 5 сек\n            chunk = raw_audio[i : i + self.chunk_len]\n            if len(chunk) < self.chunk_len:\n                padding = np.zeros(self.chunk_len - len(chunk), dtype=\"float32\")\n                chunk = np.concatenate([chunk, padding])\n            chunks.append(chunk)\n            \n        if not chunks:\n            chunks = [np.zeros(self.chunk_len, dtype=np.float32)]\n\n        images = [self._process_chunk(c) for c in chunks]\n        return np.stack(images)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:13.442359Z","iopub.execute_input":"2025-12-15T19:42:13.442696Z","iopub.status.idle":"2025-12-15T19:42:13.453247Z","shell.execute_reply.started":"2025-12-15T19:42:13.442671Z","shell.execute_reply":"2025-12-15T19:42:13.452404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds_test = InferenceDataset(df=meta_df)\nprint(f\"Dataset size: {len(ds_test)}\")\n\nfirst_item = ds_test[0]\nprint(f\"Batch shape: {first_item.shape}\")\n\nplt.figure(figsize=(8, 3))\nplt.imshow(first_item[0].transpose(1, 2, 0))\nplt.title(\"Sample Spectrogram\")\nplt.axis(\"off\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:15.114497Z","iopub.execute_input":"2025-12-15T19:42:15.115007Z","iopub.status.idle":"2025-12-15T19:42:19.781030Z","shell.execute_reply.started":"2025-12-15T19:42:15.114980Z","shell.execute_reply":"2025-12-15T19:42:19.780231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:22.780931Z","iopub.execute_input":"2025-12-15T19:42:22.781395Z","iopub.status.idle":"2025-12-15T19:42:22.973369Z","shell.execute_reply.started":"2025-12-15T19:42:22.781353Z","shell.execute_reply":"2025-12-15T19:42:22.972376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_train = pd.read_csv(BASE_DIR / \"train_metadata.csv\")\nunique_labels = sorted(meta_train[\"primary_label\"].unique())\nCLASS_TO_ID = {lbl: i for i, lbl in enumerate(unique_labels)}\nID_TO_CLASS = {i: lbl for lbl, i in CLASS_TO_ID.items()}\nprint(f\"Classes count: {len(CLASS_TO_ID)}\")\n\ndef get_model(ckpt_path):\n    model = resnest50(pretrained=False)\n    # Меняем последний слой\n    model.fc = nn.Linear(model.fc.in_features, len(CLASS_TO_ID))\n    \n    weights = torch.load(ckpt_path, map_location=\"cpu\")\n    # Убираем префикс \"model.\" если есть\n    clean_weights = {k.replace(\"model.\", \"\"): v for k, v in weights.items()}\n    \n    model.load_state_dict(clean_weights, strict=True)\n    model.to(ACCELERATOR)\n    model.eval()\n    return model\n\nmodel_path = Path(\"/kaggle/input/kkiller-birdclef-models-public/birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\")\nmodels_list = [get_model(model_path)]\nprint(\"Model loaded successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:23.968729Z","iopub.execute_input":"2025-12-15T19:42:23.969402Z","iopub.status.idle":"2025-12-15T19:42:25.075503Z","shell.execute_reply.started":"2025-12-15T19:42:23.969375Z","shell.execute_reply":"2025-12-15T19:42:25.074678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef decode_output(probs, threshold=CONF_THRESH):\n    # probs: tensor [batch, n_classes]\n    # Возвращаем имена птиц\n    results = []\n    for row in probs:\n        indices = torch.where(row > threshold)[0]\n        if len(indices) == 0:\n            results.append(\"nocall\")\n        else:\n            names = [ID_TO_CLASS[x.item()] for x in indices]\n            # Сортируем по вероятности (опционально), здесь просто список\n            results.append(\" \".join(names))\n    return results\n\n@torch.no_grad()\ndef inference_loop(models, dataset, batch_size=32):\n    final_preds = []\n    \n    for i in tqdm(range(len(dataset)), desc=\"Processing files\"):\n        file_imgs = dataset[i] # [num_windows, 3, H, W]\n        n_windows = file_imgs.shape[0]\n        file_predictions = []\n\n        for start in range(0, n_windows, batch_size):\n            end = min(start + batch_size, n_windows)\n            input_tensor = torch.tensor(file_imgs[start:end]).to(ACCELERATOR)\n            \n            # Ансамбль (здесь одна модель, но код готов к нескольким)\n            avg_preds = torch.zeros((end-start, len(CLASS_TO_ID)), device=ACCELERATOR)\n            for m in models:\n                logits = m(input_tensor)\n                avg_preds += torch.sigmoid(logits)\n            \n            avg_preds /= len(models)\n            file_predictions.extend(decode_output(avg_preds))\n        \n        final_preds.append(file_predictions)\n        \n    return final_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:29.154748Z","iopub.execute_input":"2025-12-15T19:42:29.155060Z","iopub.status.idle":"2025-12-15T19:42:29.162420Z","shell.execute_reply.started":"2025-12-15T19:42:29.155037Z","shell.execute_reply":"2025-12-15T19:42:29.161698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_predictions = inference_loop(models_list, ds_test)\nprint(f\"Processed {len(all_predictions)} files.\")\nprint(f\"First 5 windows of file 1: {all_predictions[0][:5]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:42:33.850673Z","iopub.execute_input":"2025-12-15T19:42:33.850999Z","iopub.status.idle":"2025-12-15T19:43:30.224479Z","shell.execute_reply.started":"2025-12-15T19:42:33.850976Z","shell.execute_reply":"2025-12-15T19:43:30.223612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_submission(meta, preds, step=CLIP_LEN):\n    res = []\n    for row, p_list in zip(meta.itertuples(), preds):\n        base_id = row.id\n        site = row.site\n        for i, label in enumerate(p_list):\n            seconds = (i + 1) * step\n            row_id = f\"{base_id}_{site}_{seconds}\"\n            res.append({\"row_id\": row_id, \"birds\": label})\n    \n    df_res = pd.DataFrame(res)\n    \n    # Мердж с сэмплом если он есть\n    if SUB_FILE:\n        sample = pd.read_csv(SUB_FILE, usecols=[\"row_id\"])\n        df_res = sample.merge(df_res, on=\"row_id\", how=\"left\").fillna(\"nocall\")\n        \n    return df_res\n\nsubmission_df = make_submission(meta_df, all_predictions)\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:43:32.504130Z","iopub.execute_input":"2025-12-15T19:43:32.504956Z","iopub.status.idle":"2025-12-15T19:43:32.518957Z","shell.execute_reply.started":"2025-12-15T19:43:32.504929Z","shell.execute_reply":"2025-12-15T19:43:32.518220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)\nprint(\"File saved: submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:43:35.075022Z","iopub.execute_input":"2025-12-15T19:43:35.075331Z","iopub.status.idle":"2025-12-15T19:43:35.085222Z","shell.execute_reply.started":"2025-12-15T19:43:35.075314Z","shell.execute_reply":"2025-12-15T19:43:35.084393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calc_f1_metric(true_list, pred_list):\n    # Метрика F1 Micro row-wise\n    tp_tot, fp_tot, fn_tot = 0, 0, 0\n    \n    for t, p in zip(true_list, pred_list):\n        t = \"nocall\" if pd.isna(t) else t\n        p = \"nocall\" if pd.isna(p) else p\n        \n        set_t = set(t.split()) if t != \"nocall\" else set()\n        set_p = set(p.split()) if p != \"nocall\" else set()\n        \n        tp = len(set_t & set_p)\n        fp = len(set_p - set_t)\n        fn = len(set_t - set_p)\n        \n        tp_tot += tp\n        fp_tot += fp\n        fn_tot += fn\n\n    if (tp_tot + fp_tot + fn_tot) == 0: return 0.0\n    \n    prec = tp_tot / (tp_tot + fp_tot + 1e-9)\n    rec = tp_tot / (tp_tot + fn_tot + 1e-9)\n    \n    if (prec + rec) == 0: return 0.0\n    return 2 * prec * rec / (prec + rec)\n\nif LABELS_FILE:\n    print(\"Calculating offline score...\")\n    true_df = pd.read_csv(LABELS_FILE)\n    combined = true_df.merge(submission_df, on=\"row_id\", suffixes=(\"_real\", \"_pred\"))\n    \n    metric = calc_f1_metric(combined[\"birds_real\"], combined[\"birds_pred\"])\n    print(f\"Validation F1 Score: {metric:.4f}\")\nelse:\n    print(\"Submission mode. Score available on Leaderboard.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T19:43:36.882169Z","iopub.execute_input":"2025-12-15T19:43:36.883075Z","iopub.status.idle":"2025-12-15T19:43:36.901938Z","shell.execute_reply.started":"2025-12-15T19:43:36.883046Z","shell.execute_reply":"2025-12-15T19:43:36.901111Z"}},"outputs":[],"execution_count":null}]}