{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Установка пользовательского пакета\nimport shutil\nimport os\nshutil.copytree('../input/resnest50-fast-package/resnest-0.0.6b20200701/resnest', 'resnet', dirs_exist_ok=True)\nos.system('pip install \"./resnet\" --no-deps')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:56.541960Z","iopub.execute_input":"2025-11-16T13:45:56.542864Z","iopub.status.idle":"2025-11-16T13:45:59.891240Z","shell.execute_reply.started":"2025-11-16T13:45:56.542832Z","shell.execute_reply":"2025-11-16T13:45:59.890546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Imports\nimport pandas as pd\nimport numpy as np\nimport librosa as lb\nimport soundfile as sf\nimport cv2\nfrom pathlib import Path\nimport re\nimport torch\nfrom torch import nn\nfrom  torch.utils.data import Dataset, DataLoader\nfrom tqdm.notebook import tqdm\nimport time\nfrom resnest.torch import resnest50\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.892451Z","iopub.execute_input":"2025-11-16T13:45:59.892661Z","iopub.status.idle":"2025-11-16T13:45:59.897274Z","shell.execute_reply.started":"2025-11-16T13:45:59.892644Z","shell.execute_reply":"2025-11-16T13:45:59.896515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.898231Z","iopub.execute_input":"2025-11-16T13:45:59.898485Z","iopub.status.idle":"2025-11-16T13:45:59.916686Z","shell.execute_reply.started":"2025-11-16T13:45:59.898463Z","shell.execute_reply":"2025-11-16T13:45:59.915998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Переменные для вычислений\nNUM_CLASSES = 397\nSR = 32_000\nDURATION = 5\nTHRESH = 0.21","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.918192Z","iopub.execute_input":"2025-11-16T13:45:59.918851Z","iopub.status.idle":"2025-11-16T13:45:59.930921Z","shell.execute_reply.started":"2025-11-16T13:45:59.918825Z","shell.execute_reply":"2025-11-16T13:45:59.930180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Пути к файлам\nTEST_AUDIO_ROOT = Path(\"../input/birdclef-2021/test_soundscapes\")\nSAMPLE_SUB_PATH = \"../input/birdclef-2021/sample_submission.csv\"\nTARGET_PATH = None\nif not len(list(TEST_AUDIO_ROOT.glob(\"*.ogg\"))):\n    TEST_AUDIO_ROOT = Path(\"../input/birdclef-2021/train_soundscapes\")\n    SAMPLE_SUB_PATH = None\n    # SAMPLE_SUB_PATH = \"../input/birdclef-2021/sample_submission.csv\"\n    TARGET_PATH = Path(\"../input/birdclef-2021/train_soundscape_labels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.931749Z","iopub.execute_input":"2025-11-16T13:45:59.932018Z","iopub.status.idle":"2025-11-16T13:45:59.946900Z","shell.execute_reply.started":"2025-11-16T13:45:59.931999Z","shell.execute_reply":"2025-11-16T13:45:59.946098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Вычисление мел-спектрограммы аудиодорожки. \nclass MelSpecComputer:\n    def __init__(self, sr, n_mels, fmin, fmax, **kwargs):\n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax\n        kwargs[\"n_fft\"] = kwargs.get(\"n_fft\", self.sr//10)\n        kwargs[\"hop_length\"] = kwargs.get(\"hop_length\", self.sr//(10*4))\n        self.kwargs = kwargs\n\n    def __call__(self, y):\n\n        melspec = lb.feature.melspectrogram(\n            y=y, sr=self.sr, n_mels=self.n_mels, fmin=self.fmin, fmax=self.fmax, **self.kwargs,\n        )\n\n        melspec = lb.power_to_db(melspec).astype(np.float32)\n        return melspec\n\n# Функция mono_to_color преобразует монохромное изображение в цветное. Принимает на вход массив значений, считает среднее и стандартное отклонение, затем нормализует значения массива и масштабирует их до диапазона 0-255.\ndef mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\n#Функция crop_or_pad обрезает или дополняет аудиодорожку до заданной длины\ndef crop_or_pad(y, length):\n    if len(y) < length:\n        y = np.concatenate([y, length - np.zeros(len(y))])\n    elif len(y) > length:\n        y = y[:length]\n    return y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.947778Z","iopub.execute_input":"2025-11-16T13:45:59.948024Z","iopub.status.idle":"2025-11-16T13:45:59.956875Z","shell.execute_reply.started":"2025-11-16T13:45:59.948003Z","shell.execute_reply":"2025-11-16T13:45:59.956123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для визуализации waveform\ndef plot_waveform(audio, sr, title=\"Waveform\"):\n    plt.figure(figsize=(12, 4))\n    lb.display.waveshow(audio, sr=sr)\n    plt.title(title)\n    plt.xlabel(\"Time (s)\")\n    plt.ylabel(\"Amplitude\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.957711Z","iopub.execute_input":"2025-11-16T13:45:59.957956Z","iopub.status.idle":"2025-11-16T13:45:59.972730Z","shell.execute_reply.started":"2025-11-16T13:45:59.957927Z","shell.execute_reply":"2025-11-16T13:45:59.971911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для визуализации mel-спектрограмм\ndef plot_mel_spectrogram(melspec, sr, hop_length, n_mels, title=\"Mel Spectrogram\"):\n    plt.figure(figsize=(12, 6))\n    lb.display.specshow(melspec, sr=sr, hop_length=hop_length, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(title)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.973315Z","iopub.execute_input":"2025-11-16T13:45:59.973520Z","iopub.status.idle":"2025-11-16T13:45:59.987239Z","shell.execute_reply.started":"2025-11-16T13:45:59.973500Z","shell.execute_reply":"2025-11-16T13:45:59.986447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для прослушивания аудио\ndef play_audio(audio, sr, title=\"Audio\"):\n    print(f\"Playing: {title}\")\n    display(ipd.Audio(audio, rate=sr))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:45:59.987923Z","iopub.execute_input":"2025-11-16T13:45:59.988103Z","iopub.status.idle":"2025-11-16T13:45:59.998889Z","shell.execute_reply.started":"2025-11-16T13:45:59.988089Z","shell.execute_reply":"2025-11-16T13:45:59.998164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для выполнения визуализации и прослушивания\ndef process_and_visualize(dataset, num_samples=2):\n    for i in tqdm(range(min(len(dataset), num_samples)), desc=\"Processing and visualizing samples\"):\n        filepath = dataset.data.loc[i, \"filepath\"]\n        original_audio, sr = sf.read(filepath, dtype=\"float32\")\n\n        if dataset.resample and sr != dataset.sr:\n            original_audio = lb.resample(original_audio, sr, dataset.sr, res_type=dataset.res_type)\n        \n        # Прослушивание аудио\n        play_audio(original_audio, dataset.sr, title=f\"Sample {i+1}: {dataset.data.loc[i, 'filename']}\")\n        \n        # Визуализация waveform\n        plot_waveform(original_audio, dataset.sr, title=f\"Waveform for {dataset.data.loc[i, 'filename']}\")\n        \n        # Вычисление mel-спектрограммы для визуализации\n        melspec_for_plot = dataset.mel_spec_computer(original_audio)\n        # Визуализация mel-спектрограммы\n        plot_mel_spectrogram(melspec_for_plot, dataset.sr, dataset.mel_spec_computer.kwargs[\"hop_length\"], dataset.n_mels, title=f\"Mel Spectrogram for {dataset.data.loc[i, 'filename']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:00.001120Z","iopub.execute_input":"2025-11-16T13:46:00.001393Z","iopub.status.idle":"2025-11-16T13:46:00.012746Z","shell.execute_reply.started":"2025-11-16T13:46:00.001373Z","shell.execute_reply":"2025-11-16T13:46:00.012028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обработка аудиоданных птичьих голосов\nclass BirdCLEFDataset(Dataset):\n    def __init__(self, data, sr=SR, n_mels=128, fmin=0, fmax=None, duration=DURATION, step=None, res_type=\"kaiser_fast\", resample=True):\n        \n        self.data = data\n        \n        self.sr = sr\n        self.n_mels = n_mels\n        self.fmin = fmin\n        self.fmax = fmax or self.sr//2\n\n        self.duration = duration\n        self.audio_length = self.duration*self.sr\n        self.step = step or self.audio_length\n        \n        self.res_type = res_type\n        self.resample = resample\n\n        self.mel_spec_computer = MelSpecComputer(sr=self.sr, n_mels=self.n_mels, fmin=self.fmin,\n                                                 fmax=self.fmax)\n    def __len__(self):\n        return len(self.data)\n    \n    @staticmethod\n    def normalize(image):\n        image = image.astype(\"float32\", copy=False) / 255.0\n        image = np.stack([image, image, image])\n        return image\n    \n    def audio_to_image(self, audio):\n        melspec = self.mel_spec_computer(audio) \n        image = mono_to_color(melspec)\n        image = self.normalize(image)\n        return image\n\n    def read_file(self, filepath):\n        audio, orig_sr = sf.read(filepath, dtype=\"float32\")\n\n        if self.resample and orig_sr != self.sr:\n            audio = lb.resample(audio, orig_sr, self.sr, res_type=self.res_type)\n          \n        audios = []\n        for i in range(self.audio_length, len(audio) + self.step, self.step):\n            start = max(0, i - self.audio_length)\n            end = start + self.audio_length\n            audios.append(audio[start:end])\n            \n        if len(audios[-1]) < self.audio_length:\n            audios = audios[:-1]\n            \n        images = [self.audio_to_image(audio) for audio in audios]\n        images = np.stack(images)\n        \n        return images\n    \n        \n    def __getitem__(self, idx):\n        return self.read_file(self.data.loc[idx, \"filepath\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:00.013303Z","iopub.execute_input":"2025-11-16T13:46:00.013504Z","iopub.status.idle":"2025-11-16T13:46:00.026849Z","shell.execute_reply.started":"2025-11-16T13:46:00.013490Z","shell.execute_reply":"2025-11-16T13:46:00.026246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.DataFrame(\n     [(path.stem, *path.stem.split(\"_\"), path) for path in Path(TEST_AUDIO_ROOT).glob(\"*.ogg\")],\n    columns = [\"filename\", \"id\", \"site\", \"date\", \"filepath\"]\n)\nprint(data.shape)\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:00.027535Z","iopub.execute_input":"2025-11-16T13:46:00.027775Z","iopub.status.idle":"2025-11-16T13:46:00.052275Z","shell.execute_reply.started":"2025-11-16T13:46:00.027752Z","shell.execute_reply":"2025-11-16T13:46:00.051346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обучающие данные и информация о лейблах\ndf_train = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\n\nLABEL_IDS = {label: label_id for label_id,label in enumerate(sorted(df_train[\"primary_label\"].unique()))}\nINV_LABEL_IDS = {val: key for key,val in LABEL_IDS.items()}\n\ntest_data = BirdCLEFDataset(data=data)\nlen(test_data), test_data[0].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:00.052979Z","iopub.execute_input":"2025-11-16T13:46:00.053238Z","iopub.status.idle":"2025-11-16T13:46:02.454609Z","shell.execute_reply.started":"2025-11-16T13:46:00.053188Z","shell.execute_reply":"2025-11-16T13:46:02.453859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для визуализации\nprocess_and_visualize(test_data, num_samples=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:02.455345Z","iopub.execute_input":"2025-11-16T13:46:02.455544Z","iopub.status.idle":"2025-11-16T13:46:21.058688Z","shell.execute_reply.started":"2025-11-16T13:46:02.455529Z","shell.execute_reply":"2025-11-16T13:46:21.057930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Загрузка предобученной модели\ndef load_net(checkpoint_path, num_classes=NUM_CLASSES):\n    net = resnest50(pretrained=False)\n    net.fc = nn.Linear(net.fc.in_features, num_classes)\n    dummy_device = torch.device(\"cpu\")\n    d = torch.load(checkpoint_path, map_location=dummy_device)\n    for key in list(d.keys()):\n        d[key.replace(\"model.\", \"\")] = d.pop(key)\n    net.load_state_dict(d)\n    net = net.to(DEVICE)\n    net = net.eval()\n    return net","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:21.059567Z","iopub.execute_input":"2025-11-16T13:46:21.059811Z","iopub.status.idle":"2025-11-16T13:46:21.064997Z","shell.execute_reply.started":"2025-11-16T13:46:21.059789Z","shell.execute_reply":"2025-11-16T13:46:21.064158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint_paths = [\n    Path(\"../input/kkiller-birdclef-models-public/birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\"),\n]\n\n\nnets = [\n        load_net(checkpoint_path.as_posix()) for checkpoint_path in checkpoint_paths\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:21.065823Z","iopub.execute_input":"2025-11-16T13:46:21.066271Z","iopub.status.idle":"2025-11-16T13:46:21.689464Z","shell.execute_reply.started":"2025-11-16T13:46:21.066249Z","shell.execute_reply":"2025-11-16T13:46:21.688833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Постобработка предсказаний\n@torch.no_grad()\ndef get_thresh_preds(out, thresh=None):\n    thresh = thresh or THRESH\n    o = (-out).argsort(1)\n    npreds = (out > thresh).sum(1)\n    preds = []\n    for oo, npred in zip(o, npreds):\n        preds.append(oo[:npred].cpu().numpy().tolist())\n    return preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:21.690509Z","iopub.execute_input":"2025-11-16T13:46:21.690733Z","iopub.status.idle":"2025-11-16T13:46:21.695026Z","shell.execute_reply.started":"2025-11-16T13:46:21.690716Z","shell.execute_reply":"2025-11-16T13:46:21.694385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_bird_names(preds):\n    bird_names = []\n    for pred in preds:\n        if not pred:\n            bird_names.append(\"nocall\")\n        else:\n            bird_names.append(\" \".join([INV_LABEL_IDS[bird_id] for bird_id in pred]))\n    return bird_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:21.695802Z","iopub.execute_input":"2025-11-16T13:46:21.696043Z","iopub.status.idle":"2025-11-16T13:46:21.707157Z","shell.execute_reply.started":"2025-11-16T13:46:21.696027Z","shell.execute_reply":"2025-11-16T13:46:21.706621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для предсказания\ndef predict(nets, test_data, names=True):\n    preds = []\n    with torch.no_grad():\n        for idx in  tqdm(list(range(len(test_data)))):\n            xb = torch.from_numpy(test_data[idx]).to(DEVICE)\n            pred = 0.\n            for net in nets:\n                o = net(xb)\n                o = torch.sigmoid(o)\n\n                pred += o\n\n            pred /= len(nets)\n            \n            if names:\n                pred = get_bird_names(get_thresh_preds(pred))\n\n            preds.append(pred)\n    return preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:21.707908Z","iopub.execute_input":"2025-11-16T13:46:21.708158Z","iopub.status.idle":"2025-11-16T13:46:21.720818Z","shell.execute_reply.started":"2025-11-16T13:46:21.708134Z","shell.execute_reply":"2025-11-16T13:46:21.720029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_probas = predict(nets, test_data, names=False)\nprint(len(pred_probas))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:46:21.721532Z","iopub.execute_input":"2025-11-16T13:46:21.721769Z","iopub.status.idle":"2025-11-16T13:47:04.747469Z","shell.execute_reply.started":"2025-11-16T13:46:21.721749Z","shell.execute_reply":"2025-11-16T13:47:04.746670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = [get_bird_names(get_thresh_preds(pred, thresh=THRESH)) for pred in pred_probas]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:47:04.748272Z","iopub.execute_input":"2025-11-16T13:47:04.748480Z","iopub.status.idle":"2025-11-16T13:47:05.080849Z","shell.execute_reply.started":"2025-11-16T13:47:04.748464Z","shell.execute_reply":"2025-11-16T13:47:05.080267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preds_as_df(data, preds):\n    sub = {\n        \"row_id\": [],\n        \"birds\": [],\n    }\n    \n    for row, pred in zip(data.itertuples(False), preds):\n        row_id = [f\"{row.id}_{row.site}_{5*i}\" for i in range(1, len(pred)+1)]\n        sub[\"birds\"] += pred\n        sub[\"row_id\"] += row_id\n        \n    sub = pd.DataFrame(sub)\n    \n    if SAMPLE_SUB_PATH:\n        sample_sub = pd.read_csv(SAMPLE_SUB_PATH, usecols=[\"row_id\"])\n        sub = sample_sub.merge(sub, on=\"row_id\", how=\"left\")\n        sub[\"birds\"] = sub[\"birds\"].fillna(\"nocall\")\n    return sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:47:05.081526Z","iopub.execute_input":"2025-11-16T13:47:05.081723Z","iopub.status.idle":"2025-11-16T13:47:05.087478Z","shell.execute_reply.started":"2025-11-16T13:47:05.081708Z","shell.execute_reply":"2025-11-16T13:47:05.086662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = preds_as_df(data, preds)\nprint(sub.shape)\nsub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:47:05.088301Z","iopub.execute_input":"2025-11-16T13:47:05.088541Z","iopub.status.idle":"2025-11-16T13:47:05.110443Z","shell.execute_reply.started":"2025-11-16T13:47:05.088518Z","shell.execute_reply":"2025-11-16T13:47:05.109719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Сохранение результатов\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-16T13:47:05.111136Z","iopub.execute_input":"2025-11-16T13:47:05.111438Z","iopub.status.idle":"2025-11-16T13:47:05.124511Z","shell.execute_reply.started":"2025-11-16T13:47:05.111413Z","shell.execute_reply":"2025-11-16T13:47:05.123695Z"}},"outputs":[],"execution_count":null}]}