{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":1297722,"sourceType":"datasetVersion","datasetId":750498},{"sourceId":2130303,"sourceType":"datasetVersion","datasetId":1278322}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import shutil\nimport os\n#Копирование и установка ResNeSt\nshutil.copytree('../input/resnest50-fast-package/resnest-0.0.6b20200701/resnest', 'resnet', dirs_exist_ok=True)\nos.system('pip install \"./resnet\" --no-deps')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:48.726421Z","iopub.execute_input":"2025-11-15T15:15:48.726943Z","iopub.status.idle":"2025-11-15T15:15:53.453150Z","shell.execute_reply.started":"2025-11-15T15:15:48.726919Z","shell.execute_reply":"2025-11-15T15:15:53.452554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom resnest.torch import resnest50\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:53.454140Z","iopub.execute_input":"2025-11-15T15:15:53.454370Z","iopub.status.idle":"2025-11-15T15:15:57.505957Z","shell.execute_reply.started":"2025-11-15T15:15:53.454353Z","shell.execute_reply":"2025-11-15T15:15:57.505346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Глобальные константы\nAUDIO_SAMPLE_RATE = 32000\nSEGMENT_DURATION = 5  #секунд\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nDATA_ROOT = \"../input/birdclef-2021\"\nTEST_AUDIO_PATH = os.path.join(DATA_ROOT, \"test_soundscapes\")\nMODEL_WEIGHTS_FILE = \"../input/kkiller-birdclef-models-public/birdclef_resnest50_fold0_epoch_10_f1_val_06471_20210417161101.pth\"\n\nprint(f\"Device: {DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:57.506659Z","iopub.execute_input":"2025-11-15T15:15:57.507073Z","iopub.status.idle":"2025-11-15T15:15:57.565717Z","shell.execute_reply.started":"2025-11-15T15:15:57.507053Z","shell.execute_reply":"2025-11-15T15:15:57.564922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Загрузка метаданных и кодирование классов\ntrain_meta = pd.read_csv(os.path.join(DATA_ROOT, \"train_metadata.csv\"))\nspecies_list = sorted(train_meta[\"primary_label\"].unique())\nlabel_encoder = LabelEncoder().fit(species_list)\nNUM_SPECIES = len(species_list)\nprint(f\"Total species: {NUM_SPECIES}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:57.567438Z","iopub.execute_input":"2025-11-15T15:15:57.568043Z","iopub.status.idle":"2025-11-15T15:15:58.000027Z","shell.execute_reply.started":"2025-11-15T15:15:57.568012Z","shell.execute_reply":"2025-11-15T15:15:57.999179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species_list[0:20:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:58.000880Z","iopub.execute_input":"2025-11-15T15:15:58.001251Z","iopub.status.idle":"2025-11-15T15:15:58.006277Z","shell.execute_reply.started":"2025-11-15T15:15:58.001228Z","shell.execute_reply":"2025-11-15T15:15:58.005726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Преобразование аудио в нормализованную мел-спектрограмму\ndef transform_audio_to_spec(signal, sr=AUDIO_SAMPLE_RATE):\n    mel_spec = librosa.feature.melspectrogram(\n        y=signal, sr=sr, n_mels=128, fmin=0, fmax=sr//2,\n        n_fft=sr//10, hop_length=sr//40\n    )\n    log_mel = librosa.power_to_db(mel_spec, ref=np.max)\n    normalized = (log_mel - log_mel.mean()) / (log_mel.std() + 1e-8)\n\n    min_val, max_val = normalized.min(), normalized.max()\n    if max_val - min_val > 1e-6:\n        scaled = 255 * (normalized - min_val) / (max_val - min_val)\n    else:\n        scaled = np.zeros_like(normalized)\n    return scaled.astype(np.uint8)\n\n#Преобразование спектрограммы в 3-канальный тензор\ndef make_rgb_tensor(spec_img):\n    return np.stack([spec_img] * 3, axis=0).astype(np.float32) / 255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:58.006961Z","iopub.execute_input":"2025-11-15T15:15:58.007201Z","iopub.status.idle":"2025-11-15T15:15:58.021559Z","shell.execute_reply.started":"2025-11-15T15:15:58.007185Z","shell.execute_reply":"2025-11-15T15:15:58.020854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#датасет для сегментов аудио\nclass AudioSegmentDataset(Dataset):\n    def __init__(self, annotations, audio_folder, sr=AUDIO_SAMPLE_RATE, seg_len_sec=5):\n        self.annotations = annotations.reset_index(drop=True)\n        self.folder = audio_folder\n        self.sr = sr\n        self.seg_len = seg_len_sec\n        self._audio_cache = {}\n\n    def __len__(self):\n        return len(self.annotations)\n\n    def __getitem__(self, idx):\n        row = self.annotations.iloc[idx]\n        row_id = row[\"row_id\"]\n        file_prefix, _, end_time = row_id.rsplit(\"_\", 2)\n        full_prefix = \"_\".join(row_id.split(\"_\")[:2])\n\n        try:\n            if full_prefix not in self._audio_cache:\n                audio_filename = next(f for f in os.listdir(self.folder) if f.startswith(full_prefix))\n                audio_full, orig_sr = librosa.load(os.path.join(self.folder, audio_filename), sr=None, res_type='kaiser_fast')\n                if orig_sr != self.sr:\n                    audio_full = librosa.resample(audio_full, orig_sr=orig_sr, target_sr=self.sr)\n                self._audio_cache[full_prefix] = audio_full\n            else:\n                audio_full = self._audio_cache[full_prefix]\n\n            start_sample = max(0, (int(end_time) - self.seg_len) * self.sr)\n            end_sample = min(len(audio_full), int(end_time) * self.sr)\n            segment = audio_full[start_sample:end_sample]\n\n            if len(segment) < self.seg_len * self.sr:\n                segment = np.pad(segment, (0, self.seg_len * self.sr - len(segment)))\n\n            spec = transform_audio_to_spec(segment, self.sr)\n            tensor = make_rgb_tensor(spec)\n            return tensor\n\n        except Exception:\n            return np.zeros((3, 128, 313), dtype=np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:58.022621Z","iopub.execute_input":"2025-11-15T15:15:58.022947Z","iopub.status.idle":"2025-11-15T15:15:58.040031Z","shell.execute_reply.started":"2025-11-15T15:15:58.022926Z","shell.execute_reply":"2025-11-15T15:15:58.039363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Загрузка модели ResNeSt с весами\ndef build_inference_model(weight_path, num_classes):\n    net = resnest50(pretrained=False)\n    net.fc = torch.nn.Linear(net.fc.in_features, num_classes)\n\n    checkpoint = torch.load(weight_path, map_location=\"cpu\")\n    clean_state = {k.replace(\"model.\", \"\"): v for k, v in checkpoint.items()}\n    net.load_state_dict(clean_state)\n    net.to(DEVICE)\n    net.eval()\n    return net","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:58.040809Z","iopub.execute_input":"2025-11-15T15:15:58.041022Z","iopub.status.idle":"2025-11-15T15:15:58.052278Z","shell.execute_reply.started":"2025-11-15T15:15:58.041005Z","shell.execute_reply":"2025-11-15T15:15:58.051542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Функция для инференса модели\ndef run_inference_on_batch(batch_data, model, thr=0.1):\n    with torch.no_grad():\n        inputs = torch.from_numpy(batch_data).to(DEVICE)\n        logits = model(inputs)\n        probs = torch.sigmoid(logits).cpu().numpy()\n\n    results = []\n    for prob_vec in probs:\n        active_labels = np.where(prob_vec > thr)[0]\n        if len(active_labels) == 0:\n            results.append(\"nocall\")\n        else:\n            names = label_encoder.inverse_transform(active_labels)\n            results.append(\" \".join(sorted(names)))\n    return results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:15:58.053015Z","iopub.execute_input":"2025-11-15T15:15:58.053214Z","iopub.status.idle":"2025-11-15T15:15:58.068866Z","shell.execute_reply.started":"2025-11-15T15:15:58.053192Z","shell.execute_reply":"2025-11-15T15:15:58.068105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Подготовка тестовых данных\ntest_meta = pd.read_csv(os.path.join(DATA_ROOT, \"test.csv\"))\nuse_train_as_test = len(test_meta) < 10\naudio_source = os.path.join(DATA_ROOT, \"train_soundscapes\") if use_train_as_test else TEST_AUDIO_PATH\nif use_train_as_test:\n    test_meta = pd.read_csv(os.path.join(DATA_ROOT, \"train_soundscape_labels.csv\"))\n\nprint(f\"Total segments to process: {len(test_meta)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:18:02.802411Z","iopub.execute_input":"2025-11-15T15:18:02.802738Z","iopub.status.idle":"2025-11-15T15:18:02.828764Z","shell.execute_reply.started":"2025-11-15T15:18:02.802716Z","shell.execute_reply":"2025-11-15T15:18:02.827946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Загрузка данных и инференс модели\naudio_dataset = AudioSegmentDataset(test_meta, audio_source)\nloader = DataLoader(audio_dataset, batch_size=64, shuffle=False, num_workers=0)\n\nfinal_predictions = []\nfor batch in tqdm(loader, desc=\"Running inference\"):\n    stacked_batch = np.stack([b.numpy() for b in batch])\n    batch_preds = run_inference_on_batch(stacked_batch, build_inference_model(MODEL_WEIGHTS_FILE, NUM_SPECIES), thr=0.1)\n    final_predictions.extend(batch_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:18:06.652763Z","iopub.execute_input":"2025-11-15T15:18:06.653461Z","iopub.status.idle":"2025-11-15T15:19:32.896778Z","shell.execute_reply.started":"2025-11-15T15:18:06.653440Z","shell.execute_reply":"2025-11-15T15:19:32.896110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Audio, display\n\n#Несколько примеров для визуализации\ndetected_examples = [i for i, p in enumerate(final_predictions) if p != \"nocall\"][-10:]\n\nif detected_examples:\n    for idx, i in enumerate(detected_examples, 1):\n        row_id = test_meta.iloc[i][\"row_id\"]\n        pred_label = final_predictions[i]\n        print(f\"\\n{'='*60}\")\n        print(f\"Example {idx} | Row ID: {row_id}\")\n        print(f\"Predicted birds: '{pred_label}'\")\n        print(f\"{'='*60}\")\n\n        #Загрузка аудио\n        prefix_key = \"_\".join(row_id.split(\"_\")[:2])\n        end_sec = int(row_id.split(\"_\")[-1])\n\n        audio_file = next(f for f in os.listdir(audio_source) if f.startswith(prefix_key))\n        full_audio, sr_orig = librosa.load(os.path.join(audio_source, audio_file), sr=None)\n        if sr_orig != AUDIO_SAMPLE_RATE:\n            full_audio = librosa.resample(full_audio, orig_sr=sr_orig, target_sr=AUDIO_SAMPLE_RATE)\n\n        start_samp = max(0, (end_sec - SEGMENT_DURATION) * AUDIO_SAMPLE_RATE)\n        end_samp = min(len(full_audio), end_sec * AUDIO_SAMPLE_RATE)\n        audio_clip = full_audio[start_samp:end_samp]\n        if len(audio_clip) < SEGMENT_DURATION * AUDIO_SAMPLE_RATE:\n            audio_clip = np.pad(audio_clip, (0, SEGMENT_DURATION * AUDIO_SAMPLE_RATE - len(audio_clip)))\n\n        display(Audio(audio_clip, rate=AUDIO_SAMPLE_RATE))\n\n        #waveform и спектрограмма\n        fig, axs = plt.subplots(2, 1, figsize=(14, 6), gridspec_kw={'height_ratios': [1, 2]})\n\n        #waveform\n        time_wave = np.linspace(0, SEGMENT_DURATION, len(audio_clip))\n        axs[0].plot(time_wave, audio_clip, color='steelblue', linewidth=0.8)\n        axs[0].set_title(\"Waveform\", fontsize=11)\n        axs[0].set_ylabel(\"Amplitude\")\n        axs[0].set_xlim(0, SEGMENT_DURATION)\n        axs[0].grid(True, linestyle='--', alpha=0.5)\n\n        #спектрограмма\n        spec = audio_dataset[i][0] \n        im = axs[1].imshow(\n            spec,\n            aspect='auto',\n            origin='lower',\n            cmap='viridis', \n            extent=[0, SEGMENT_DURATION, 0, AUDIO_SAMPLE_RATE // 2 // 1000]  # ось частот в кГц\n        )\n        axs[1].set_title(\"Mel-spectrogram\", fontsize=11)\n        axs[1].set_xlabel(\"Время (секунды)\")\n        axs[1].set_ylabel(\"Частота (кГц)\")\n        plt.colorbar(im, ax=axs[1], shrink=0.6)\n\n        plt.tight_layout()\n        plt.show()\n\nelse:\n    print(\"Классы не определены.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:19:32.897849Z","iopub.execute_input":"2025-11-15T15:19:32.898224Z","iopub.status.idle":"2025-11-15T15:19:45.327081Z","shell.execute_reply.started":"2025-11-15T15:19:32.898205Z","shell.execute_reply":"2025-11-15T15:19:45.326094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Сохранение результата\nsubmission_df = pd.DataFrame({\n    \"row_id\": test_meta[\"row_id\"],\n    \"birds\": final_predictions\n})\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(submission_df.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-15T15:19:45.327957Z","iopub.execute_input":"2025-11-15T15:19:45.328200Z","iopub.status.idle":"2025-11-15T15:19:45.347025Z","shell.execute_reply.started":"2025-11-15T15:19:45.328183Z","shell.execute_reply":"2025-11-15T15:19:45.346099Z"}},"outputs":[],"execution_count":null}]}