{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport os\nimport shutil\nfrom pathlib import Path\nfrom dataclasses import dataclass\nfrom typing import List, Optional, Tuple\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport soundfile as sf\nimport torch\nimport torch.nn as nn\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:12.906435Z","iopub.execute_input":"2025-12-15T16:27:12.907297Z","iopub.status.idle":"2025-12-15T16:27:12.912261Z","shell.execute_reply.started":"2025-12-15T16:27:12.907258Z","shell.execute_reply":"2025-12-15T16:27:12.911407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"package_path = '../input/resnest50-fast-package/resnest-0.0.6b20200701/resnest'\nif Path(package_path).exists():\n    shutil.copytree(package_path, 'resnet', dirs_exist_ok=True)\n    sys.path.append('./resnet')\n    from resnest.torch import resnest50\nelse:\n    print(\"Пакет resnest не найден\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:12.913678Z","iopub.execute_input":"2025-12-15T16:27:12.913934Z","iopub.status.idle":"2025-12-15T16:27:12.992194Z","shell.execute_reply.started":"2025-12-15T16:27:12.913916Z","shell.execute_reply":"2025-12-15T16:27:12.991391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Конфигурация ---\n@dataclass\nclass Config:\n    num_classes: int = 397\n    sr: int = 32_000\n    duration: int = 5\n    threshold: float = 0.21\n    device: torch.device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    test_audio_dir: Path = Path(\"../input/birdclef-2021/test_soundscapes\")\n    train_metadata: Path = Path(\"../input/birdclef-2021/train_metadata.csv\")\n    checkpoint_path: Path = Path(\"../input/kkiller-birdclef-models-public/birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\")\n\ncfg = Config()\nprint(f\"Запуск на устройстве: {cfg.device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:12.993015Z","iopub.execute_input":"2025-12-15T16:27:12.993297Z","iopub.status.idle":"2025-12-15T16:27:13.057345Z","shell.execute_reply.started":"2025-12-15T16:27:12.993275Z","shell.execute_reply":"2025-12-15T16:27:13.056715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Утилиты для обработки аудио ---\nclass AudioTransform:\n    \"\"\"Класс для превращения аудиосигнала в Mel-спектрограмму (картинку)\"\"\"\n    def __init__(self, sr=32000, n_mels=128, fmin=0, fmax=None):\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax or sr // 2\n        self.n_fft = sr // 10\n        self.hop_length = sr // 40  # (10 * 4)\n\n    def audio_to_melspec(self, audio: np.array) -> np.array:\n        \"\"\"Генерирует мел-спектрограмму из сырого аудио\"\"\"\n        melspec = librosa.feature.melspectrogram(\n            y=audio, \n            sr=self.sr, \n            n_mels=self.n_mels, \n            fmin=self.fmin, \n            fmax=self.fmax, \n            n_fft=self.n_fft,\n            hop_length=self.hop_length,\n        )\n        melspec = librosa.power_to_db(melspec).astype(np.float32)\n        return melspec\n\n    def to_image(self, melspec: np.array) -> np.array:\n        \"\"\"Переводит спектрограмму в 3-канальное изображение (RGB)\"\"\"\n        mean, std = melspec.mean(), melspec.std()\n        image = (melspec - mean) / (std + 1e-6)\n        \n        _min, _max = image.min(), image.max()\n        if (_max - _min) > 1e-6:\n            image = np.clip(image, _min, _max)\n            image = 255 * (image - _min) / (_max - _min)\n        else:\n            image = np.zeros_like(image)\n            \n        image = image.astype(np.uint8)\n        image = np.stack([image, image, image]) \n        \n        return image.astype(\"float32\") / 255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:13.058916Z","iopub.execute_input":"2025-12-15T16:27:13.059127Z","iopub.status.idle":"2025-12-15T16:27:13.076181Z","shell.execute_reply.started":"2025-12-15T16:27:13.059110Z","shell.execute_reply":"2025-12-15T16:27:13.075635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Dataset ---\nclass InferenceDataset:\n    def __init__(self, file_paths, cfg: Config):\n        self.file_paths = file_paths\n        self.cfg = cfg\n        self.transformer = AudioTransform(sr=cfg.sr)\n        self.chunk_len = cfg.duration * cfg.sr\n\n    def __len__(self):\n        return len(self.file_paths)\n\n    def read_audio(self, filepath):\n        \"\"\"Читает аудио и ресемплирует если нужно\"\"\"\n        audio, orig_sr = sf.read(filepath, dtype=\"float32\")\n        if orig_sr != self.cfg.sr:\n            audio = librosa.resample(audio, orig_sr, self.cfg.sr, res_type=\"kaiser_fast\")\n        return audio\n\n    def __getitem__(self, idx):\n        path = self.file_paths[idx]\n        audio = self.read_audio(path)\n        \n        # нарезка аудио на куски по 5 секунд\n        chunks = []\n        for i in range(0, len(audio), self.chunk_len):\n            segment = audio[i : i + self.chunk_len]\n            # отбрасываем, если кусок короче 5 секунд (конец файла)\n            if len(segment) < self.chunk_len:\n                continue\n            \n            # звук в картинку\n            melspec = self.transformer.audio_to_melspec(segment)\n            image = self.transformer.to_image(melspec)\n            chunks.append(image)\n            \n        return np.stack(chunks) if chunks else np.array([])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:13.076770Z","iopub.execute_input":"2025-12-15T16:27:13.076999Z","iopub.status.idle":"2025-12-15T16:27:13.093886Z","shell.execute_reply.started":"2025-12-15T16:27:13.076971Z","shell.execute_reply":"2025-12-15T16:27:13.093230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Модель ---\ndef load_model(path: Path, device: torch.device):\n    \"\"\"Загружает архитектуру ResNeSt50 и веса\"\"\"\n    model = resnest50(pretrained=False)\n    model.fc = nn.Linear(model.fc.in_features, cfg.num_classes)\n    \n    state_dict = torch.load(path, map_location='cpu')\n    clean_state_dict = {k.replace(\"model.\", \"\"): v for k, v in state_dict.items()}\n    model.load_state_dict(clean_state_dict)\n    \n    model.to(device)\n    model.eval()\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:13.094667Z","iopub.execute_input":"2025-12-15T16:27:13.094850Z","iopub.status.idle":"2025-12-15T16:27:13.114517Z","shell.execute_reply.started":"2025-12-15T16:27:13.094835Z","shell.execute_reply":"2025-12-15T16:27:13.113517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Подготовка данных ---\ntest_files = list(cfg.test_audio_dir.glob(\"*.ogg\"))\nif not test_files:\n    print(\"Тестовые файлы не найдены, используем примеры из train...\")\n    cfg.test_audio_dir = Path(\"../input/birdclef-2021/train_soundscapes\")\n    test_files = list(cfg.test_audio_dir.glob(\"*.ogg\"))[:5]\n\nprint(f\"Найдено файлов для обработки: {len(test_files)}\")\n\ndf_meta = pd.read_csv(cfg.train_metadata)\nLABELS = sorted(df_meta[\"primary_label\"].unique())\nID_TO_BIRD = {i: label for i, label in enumerate(LABELS)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:13.115279Z","iopub.execute_input":"2025-12-15T16:27:13.115890Z","iopub.status.idle":"2025-12-15T16:27:13.478689Z","shell.execute_reply.started":"2025-12-15T16:27:13.115873Z","shell.execute_reply":"2025-12-15T16:27:13.478122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Инференс (Предсказание) ---\nmodel = load_model(cfg.checkpoint_path, cfg.device)\ndataset = InferenceDataset(test_files, cfg)\n\npredictions = []\nrow_ids = []\n\nprint(\"Начинаем классификацию...\")\nwith torch.no_grad():\n    for i in tqdm(range(len(dataset))):\n        batch_images = dataset[i] # Получение 5-сек отрезков из одного файла\n        if len(batch_images) == 0:\n            continue\n            \n        file_name = test_files[i].stem\n        file_parts = file_name.split(\"_\")\n        site = file_parts[1]\n        \n        # Перевод в тензор и отправка на GPU\n        inputs = torch.from_numpy(batch_images).to(cfg.device)\n        \n        # Прогон через модель\n        logits = model(inputs)\n        probs = torch.sigmoid(logits).cpu().numpy() # Вероятности (0-1)\n        \n        # Ответ для каждого отрезка\n        for chunk_idx, prob_vec in enumerate(probs):\n            seconds = (chunk_idx + 1) * 5\n            row_id = f\"{file_parts[0]}_{site}_{seconds}\"\n            \n            # Фильтр по порогу\n            detected_indices = np.where(prob_vec > cfg.threshold)[0]\n            \n            if len(detected_indices) > 0:\n                birds = \" \".join([ID_TO_BIRD[idx] for idx in detected_indices])\n            else:\n                birds = \"nocall\"\n                \n            row_ids.append(row_id)\n            predictions.append(birds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:13.479403Z","iopub.execute_input":"2025-12-15T16:27:13.479639Z","iopub.status.idle":"2025-12-15T16:27:41.106399Z","shell.execute_reply.started":"2025-12-15T16:27:13.479618Z","shell.execute_reply":"2025-12-15T16:27:41.105565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Сохранение ---\nsubmission = pd.DataFrame({\n    \"row_id\": row_ids,\n    \"birds\": predictions\n})\n\nsample_sub_path = Path(\"../input/birdclef-2021/sample_submission.csv\")\nif sample_sub_path.exists():\n    sample_sub = pd.read_csv(sample_sub_path)\n    submission = sample_sub.drop(\"birds\", axis=1).merge(submission, on=\"row_id\", how=\"left\")\n    submission[\"birds\"] = submission[\"birds\"].fillna(\"nocall\")\n\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"Файл submission.csv сохранен.\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T16:27:41.107296Z","iopub.execute_input":"2025-12-15T16:27:41.107674Z","iopub.status.idle":"2025-12-15T16:27:41.138758Z","shell.execute_reply.started":"2025-12-15T16:27:41.107654Z","shell.execute_reply":"2025-12-15T16:27:41.138217Z"}},"outputs":[],"execution_count":null}]}