{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2126919,"sourceType":"datasetVersion","datasetId":1274570},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import shutil\nshutil.copytree('/kaggle/input/resnest50-fast-package/resnest-0.0.6b20200701/resnest', 'resnet', dirs_exist_ok=True) \n!pip install \"./resnet\" --no-deps","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:38:49.433631Z","iopub.execute_input":"2025-12-07T11:38:49.434258Z","iopub.status.idle":"2025-12-07T11:38:53.766051Z","shell.execute_reply.started":"2025-12-07T11:38:49.434229Z","shell.execute_reply":"2025-12-07T11:38:53.765340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom resnest.torch import resnest50\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport soundfile as sf\nimport librosa as lb\nfrom torch import nn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:55:01.982205Z","iopub.execute_input":"2025-12-07T11:55:01.982463Z","iopub.status.idle":"2025-12-07T11:55:01.987026Z","shell.execute_reply.started":"2025-12-07T11:55:01.982446Z","shell.execute_reply":"2025-12-07T11:55:01.986361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nprint('Доступное устройство: {}'.format(device))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:40:21.195758Z","iopub.execute_input":"2025-12-07T11:40:21.196239Z","iopub.status.idle":"2025-12-07T11:40:21.275996Z","shell.execute_reply.started":"2025-12-07T11:40:21.196218Z","shell.execute_reply":"2025-12-07T11:40:21.275369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nclass AudioAnalysisConfig:\n    \"\"\"Конфигурация для анализа аудиоданных\"\"\"\n    \n    def __init__(self):\n        self.num_classes = 397\n        self.sample_rate = 32000\n        self.segment_duration = 5\n        self.threshold = 0.25\n        \n        # Определение путей к данным\n        self._setup_paths()\n    \n    def _setup_paths(self):\n        \"\"\"Настройка путей к аудиофайлам и меткам\"\"\"\n        test_root = Path(\"../input/birdclef-2021/test_soundscapes\")\n        train_root = Path(\"../input/birdclef-2021/train_soundscapes\")\n        \n        if self._has_audio_files(test_root):\n            self.audio_directory = test_root\n            self.submission_template = \"../input/birdclef-2021/sample_submission.csv\"\n            self.ground_truth = None\n        else:\n            self.audio_directory = train_root\n            self.submission_template = None\n            self.ground_truth = Path(\"../input/birdclef-2021/train_soundscape_labels.csv\")\n    \n    @staticmethod\n    def _has_audio_files(directory, extension=\"*.ogg\"):\n        \"\"\"Проверка наличия аудиофайлов в директории\"\"\"\n        return len(list(directory.glob(extension))) > 0\n\n# Использование\nconfig = AudioAnalysisConfig()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:44:46.432890Z","iopub.execute_input":"2025-12-07T11:44:46.433481Z","iopub.status.idle":"2025-12-07T11:44:46.443181Z","shell.execute_reply.started":"2025-12-07T11:44:46.433459Z","shell.execute_reply":"2025-12-07T11:44:46.442627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MelSpecComputer:\n\n    def __init__(self, sr, n_mels, fmin, fmax, **kwargs):\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax\n\n        kwargs[\"n_fft\"] = kwargs.get(\"n_fft\", self.sr//10)\n        kwargs[\"hop_length\"] = kwargs.get(\"hop_length\", self.sr//(10*4))\n\n        self.kwargs = kwargs\n\n    def __call__(self, y):\n        melspec = lb.feature.melspectrogram(y=y, sr=self.sr, n_mels=self.n_mels, fmin=self.fmin, fmax=self.fmax, **self.kwargs)\n        melspec = lb.power_to_db(melspec).astype(np.float32)\n\n        return melspec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:45:01.297369Z","iopub.execute_input":"2025-12-07T11:45:01.298184Z","iopub.status.idle":"2025-12-07T11:45:01.303115Z","shell.execute_reply.started":"2025-12-07T11:45:01.298156Z","shell.execute_reply":"2025-12-07T11:45:01.302341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_spectrogram_to_rgb(spectrogram, epsilon=1e-6, mean_val=None, std_val=None):\n    \"\"\"\n    Преобразует монохромный спектрограммный массив в цветное RGB-подобное представление.\n    \n    Параметры:\n    -----------\n    spectrogram : np.ndarray\n        Входной монохромный спектрограммный массив\n    epsilon : float, optional\n        Малое значение для избежания деления на ноль (по умолчанию 1e-6)\n    mean_val : float, optional\n        Предварительно вычисленное среднее значение для нормализации\n    std_val : float, optional\n        Предварительно вычисленное стандартное отклонение для нормализации\n    \n    Возвращает:\n    -----------\n    np.ndarray\n        Массив в формате uint8 (0-255), готовый для визуализации\n    \"\"\"\n    \n    # Нормализация данных\n    normalization_mean = mean_val if mean_val is not None else spectrogram.mean()\n    normalization_std = std_val if std_val is not None else spectrogram.std()\n    \n    normalized_data = (spectrogram - normalization_mean) / (normalization_std + epsilon)\n    \n    # Определение диапазона значений\n    data_minimum = normalized_data.min()\n    data_maximum = normalized_data.max()\n    value_range = data_maximum - data_minimum\n    \n    # Преобразование в диапазон 0-255\n    if value_range > epsilon:\n        # Масштабирование с ограничением\n        scaled_values = np.clip(normalized_data, data_minimum, data_maximum)\n        rgb_values = 255 * (scaled_values - data_minimum) / value_range\n        rgb_values = rgb_values.astype(np.uint8)\n    else:\n        # Для плоских спектрограмм возвращаем нулевой массив\n        rgb_values = np.zeros_like(normalized_data, dtype=np.uint8)\n    \n    return rgb_values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:46:03.129056Z","iopub.execute_input":"2025-12-07T11:46:03.129346Z","iopub.status.idle":"2025-12-07T11:46:03.135044Z","shell.execute_reply.started":"2025-12-07T11:46:03.129324Z","shell.execute_reply":"2025-12-07T11:46:03.134383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adjust_audio_length(audio_signal, target_length):\n    \"\"\"\n    Обрезает или дополняет аудиосигнал до заданной длины.\n    \n    Параметры:\n    ----------\n    audio_signal : np.ndarray\n        Входной аудиосигнал\n    target_length : int\n        Желаемая длина сигнала\n    \n    Возвращает:\n    ----------\n    np.ndarray\n        Сигнал заданной длины\n    \"\"\"\n    current_length = len(audio_signal)\n    \n    if current_length < target_length:\n        # Дополнение нулями\n        padding_length = target_length - current_length\n        padding = np.zeros(padding_length, dtype=audio_signal.dtype)\n        result = np.concatenate([audio_signal, padding])\n    elif current_length > target_length:\n        # Обрезка до нужной длины\n        result = audio_signal[:target_length]\n    else:\n        # Длина уже правильная\n        result = audio_signal\n    \n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:46:44.520214Z","iopub.execute_input":"2025-12-07T11:46:44.520792Z","iopub.status.idle":"2025-12-07T11:46:44.525305Z","shell.execute_reply.started":"2025-12-07T11:46:44.520766Z","shell.execute_reply":"2025-12-07T11:46:44.524564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFDataset(Dataset):\n\n    def __init__(self, data, sr=config.sample_rate, n_mels=128, fmin=0, fmax=None, duration=config.segment_duration, step=None, res_type=\"kaiser_fast\", resample=True):\n        self.data = data\n\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax or self.sr//2\n\n        self.duration = duration\n        self.audio_length = self.duration*self.sr\n        self.step = step or self.audio_length\n\n        self.res_type = res_type\n        self.resample = resample\n\n        self.mel_spec_computer = MelSpecComputer(sr=self.sr, n_mels=self.n_mels, fmin=self.fmin, fmax=self.fmax)\n\n    def __len__(self):\n        return len(self.data)\n\n    @staticmethod\n    def normalize(image):\n        image = image.astype(\"float32\", copy=False) / 255.0\n        image = np.stack([image, image, image])\n\n        return image\n\n    def audio_to_image(self, audio):\n        melspec = self.mel_spec_computer(audio) \n        image = convert_spectrogram_to_rgb(melspec)\n        image = self.normalize(image)\n\n        return image\n\n    def read_file(self, filepath):\n        audio, orig_sr = sf.read(filepath, dtype=\"float32\")\n\n        if self.resample and orig_sr != self.sr:\n            audio = lb.resample(audio, orig_sr, self.sr, res_type=self.res_type)\n\n        audios = []\n        for i in range(self.audio_length, len(audio) + self.step, self.step):\n            start = max(0, i - self.audio_length)\n            end = start + self.audio_length\n            audios.append(audio[start:end])\n\n        if len(audios[-1]) < self.audio_length:\n            audios = audios[:-1]\n\n        images = [self.audio_to_image(audio) for audio in audios]\n        images = np.stack(images)\n\n        return images\n\n    def __getitem__(self, idx):\n        return self.read_file(self.data.loc[idx, \"filepath\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:53:20.206155Z","iopub.execute_input":"2025-12-07T11:53:20.206719Z","iopub.status.idle":"2025-12-07T11:53:20.215831Z","shell.execute_reply.started":"2025-12-07T11:53:20.206694Z","shell.execute_reply":"2025-12-07T11:53:20.215011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.DataFrame(\n     [(path.stem, *path.stem.split(\"_\"), path) for path in config.audio_directory.glob(\"*.ogg\")],\n    columns = [\"filename\", \"id\", \"site\", \"date\", \"filepath\"]\n)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:49:48.124291Z","iopub.execute_input":"2025-12-07T11:49:48.125037Z","iopub.status.idle":"2025-12-07T11:49:48.155502Z","shell.execute_reply.started":"2025-12-07T11:49:48.125011Z","shell.execute_reply":"2025-12-07T11:49:48.154800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\n\nLABEL_IDS = {label: label_id for label_id,label in enumerate(sorted(df_train[\"primary_label\"].unique()))}\nINV_LABEL_IDS = {val: key for key,val in LABEL_IDS.items()}\n\ntest_data = BirdCLEFDataset(data=df)\nlen(test_data), test_data[0].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:53:23.016703Z","iopub.execute_input":"2025-12-07T11:53:23.016949Z","iopub.status.idle":"2025-12-07T11:53:25.430163Z","shell.execute_reply.started":"2025-12-07T11:53:23.016933Z","shell.execute_reply":"2025-12-07T11:53:25.429559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_net(checkpoint_path, num_classes=config.num_classes):\n    net = resnest50(pretrained=False)\n    net.fc = nn.Linear(net.fc.in_features, num_classes)\n\n    dummy_device = torch.device(\"cpu\")\n    d = torch.load(checkpoint_path, map_location=dummy_device)\n\n    for key in list(d.keys()):\n        d[key.replace(\"model.\", \"\")] = d.pop(key)\n\n    net.load_state_dict(d)\n    net = net.to(device)\n    net = net.eval()\n\n    return net","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:54:13.272892Z","iopub.execute_input":"2025-12-07T11:54:13.273644Z","iopub.status.idle":"2025-12-07T11:54:13.279431Z","shell.execute_reply.started":"2025-12-07T11:54:13.273614Z","shell.execute_reply":"2025-12-07T11:54:13.278734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef get_thresh_preds(out, thresh=None):\n    thresh = thresh or THRESH\n    o = (-out).argsort(1)\n    npreds = (out > thresh).sum(1)\n    preds = []\n\n    for oo, npred in zip(o, npreds):\n        preds.append(oo[:npred].cpu().numpy().tolist())\n\n    return preds\n\ndef get_bird_names(preds):\n    bird_names = []\n\n    for pred in preds:\n        if not pred:\n            bird_names.append(\"nocall\")\n        else:\n            bird_names.append(\" \".join([INV_LABEL_IDS[bird_id] for bird_id in pred]))\n\n    return bird_names\n\ndef predict(nets, test_data, names=True):\n    preds = []\n\n    with torch.no_grad():\n        for idx in  tqdm(list(range(len(test_data)))):\n            xb = torch.from_numpy(test_data[idx]).to(device)\n            pred = 0.\n\n            for net in nets:\n                o = net(xb)\n                o = torch.sigmoid(o)\n                pred += o\n\n            pred /= len(nets)\n\n            if names:\n                pred = get_bird_names(get_thresh_preds(pred))\n\n            preds.append(pred)\n\n    return preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:54:26.904150Z","iopub.execute_input":"2025-12-07T11:54:26.904852Z","iopub.status.idle":"2025-12-07T11:54:26.914578Z","shell.execute_reply.started":"2025-12-07T11:54:26.904818Z","shell.execute_reply":"2025-12-07T11:54:26.913919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint_paths = [Path(\"../input/kkiller-birdclef-models-public/birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\")]\n\nnets = [load_net(checkpoint_path.as_posix()) for checkpoint_path in checkpoint_paths]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T11:55:06.782620Z","iopub.execute_input":"2025-12-07T11:55:06.782882Z","iopub.status.idle":"2025-12-07T11:55:09.114289Z","shell.execute_reply.started":"2025-12-07T11:55:06.782865Z","shell.execute_reply":"2025-12-07T11:55:09.113464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_probas = predict(nets, test_data, names=False)\npreds = [get_bird_names(get_thresh_preds(pred, thresh=config.threshold)) for pred in pred_probas]\n\ndef preds_as_df(data, preds):\n    sub = {\n        \"row_id\": [],\n        \"birds\": []\n    }\n\n    for row, pred in zip(data.itertuples(False), preds):\n        row_id = [f\"{row.id}_{row.site}_{5*i}\" for i in range(1, len(pred)+1)]\n        sub[\"birds\"] += pred\n        sub[\"row_id\"] += row_id\n\n    sub = pd.DataFrame(sub)\n\n    if config.submission_template:\n        sample_sub = pd.read_csv(config.submission_template, usecols=[\"row_id\"])\n        sub = sample_sub.merge(sub, on=\"row_id\", how=\"left\")\n        sub[\"birds\"] = sub[\"birds\"].fillna(\"nocall\")\n\n    return sub\n\nsub = preds_as_df(df, preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:00:32.796652Z","iopub.execute_input":"2025-12-07T12:00:32.796935Z","iopub.status.idle":"2025-12-07T12:01:17.292425Z","shell.execute_reply.started":"2025-12-07T12:00:32.796914Z","shell.execute_reply":"2025-12-07T12:01:17.291836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-07T12:01:27.197715Z","iopub.execute_input":"2025-12-07T12:01:27.197992Z","iopub.status.idle":"2025-12-07T12:01:27.208625Z","shell.execute_reply.started":"2025-12-07T12:01:27.197973Z","shell.execute_reply":"2025-12-07T12:01:27.207979Z"}},"outputs":[],"execution_count":null}]}