{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":848739,"sourceType":"datasetVersion","datasetId":251095}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Библиотеки","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision.transforms import v2 as T # Используем transforms v2 для SpecAugment\nimport torchaudio.transforms as AT \nimport pathlib\nimport timm # Библиотека для моделей, включая EfficientNet\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm.notebook import tqdm # Для отображения прогресса\nimport warnings\nimport random\n\n# Игнорировать предупреждения от librosa (например, об audiocore)\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-01T00:50:17.867670Z","iopub.execute_input":"2025-05-01T00:50:17.867967Z","iopub.status.idle":"2025-05-01T00:50:34.429660Z","shell.execute_reply.started":"2025-05-01T00:50:17.867945Z","shell.execute_reply":"2025-05-01T00:50:34.428532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Конфигурация ","metadata":{}},{"cell_type":"code","source":"# Определение основных параметров и путей\nclass Config:\n    SEED = 42 # Для воспроизводимости\n    SAMPLE_RATE = 32000 # Целевая частота дискретизации [14]\n    DURATION_SECONDS = 5 # Длительность сегмента для обучения и предсказания [20]\n    N_MELS = 64 # Количество мел-полос [56] - может потребовать настройки\n    FMIN = 50 # Минимальная частота для мел-спектрограммы [56] - может потребовать настройки\n    FMAX = 15000 # Максимальная частота для мел-спектрограммы [56] - может потребовать настройки\n    N_FFT = 2048 # Размер окна БПФ [56] - может потребовать настройки\n    HOP_LENGTH = 512 # Шаг окна БПФ [56] - может потребовать настройки\n    N_CLASSES = 206 # Количество видов в соревновании [20]\n    MODEL_NAME = 'tf_efficientnet_b0' # Эффективная модель [6, 80] (_ns - noisy student)\n    MODEL_PATH = '../input/efficientnet-pytorch/efficientnet-b0-08094119.pth' \n    PRETRAINED = False # Использовать предобученные веса ImageNet [6, 60]\n    BATCH_SIZE = 64\n    INFERENCE_BATCH_SIZE = 64 # Can be larger than training batch size\n    LEARNING_RATE = 1e-3\n    EPOCHS = 3 # Количество эпох (может потребоваться больше)\n    NUM_WORKERS = 2 # Количество потоков для загрузки данных\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    # Пути\n    DATA_DIR = \"/kaggle/input/birdclef-2025\"\n    SAMPLE_SUBMISSION_PATH = os.path.join(DATA_DIR, \"sample_submission.csv\")\n    TEST_SOUNDSCAPES_DIR = os.path.join(DATA_DIR, \"test_soundscapes\")\n    SUBMISSION_CSV_PATH = \"submission.csv\"\n    Taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    TRAIN_AUDIO_DIR = os.path.join(DATA_DIR, \"train_audio\")\n    TRAIN_METADATA_PATH = os.path.join(DATA_DIR, \"train.csv\")\n\n\n    # Параметры аугментации\n    MIXUP_ALPHA = 0.3 \n    USE_MIXUP = True\n    USE_SPECAUGMENT = True \n\n# Функция для установки seed для воспроизводимости\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True # Можно установить в False для полной детерминированности, но медленнее\n\nseed_everything(Config.SEED)\n\n# --- Загрузка таксономии ---\nprint(\"Loading taxonomy data...\")\ntry:\n    taxonomy_df = pd.read_csv(Config.Taxonomy_csv)\n    species_ids = taxonomy_df['primary_label'].tolist()\n    num_classes = len(species_ids)\n    print(f\"Number of classes: {num_classes}\")\nexcept FileNotFoundError:\n    print(f\"Ошибка: Файл таксономии не найден по пути {cfg.taxonomy_csv}\")\n    # Можно попробовать загрузить из sample_submission, если таксономии нет\n    try:\n        sample_sub = pd.read_csv(cfg.submission_csv)\n        species_ids = sample_sub.columns[1:].tolist()\n        num_classes = len(species_ids)\n        print(f\"Загружены классы из sample_submission. Количество классов: {num_classes}\")\n    except FileNotFoundError:\n        print(f\"Ошибка: Sample submission не найден по пути {cfg.submission_csv}. Невозможно определить классы.\")\n        exit() # Выход, если классы определить не удалось\nexcept Exception as e:\n    print(f\"Ошибка при загрузке таксономии: {e}\")\n    exit()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T00:50:34.431283Z","iopub.execute_input":"2025-05-01T00:50:34.431825Z","iopub.status.idle":"2025-05-01T00:50:34.472224Z","shell.execute_reply.started":"2025-05-01T00:50:34.431797Z","shell.execute_reply":"2025-05-01T00:50:34.471417Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Подготовка данных","metadata":{}},{"cell_type":"markdown","source":"## Загрузка метаданных","metadata":{}},{"cell_type":"code","source":"try:\n    train_df = pd.read_csv(Config.TRAIN_METADATA_PATH)\n    print(f\"Загружено {len(train_df)} записей из {Config.TRAIN_METADATA_PATH}\")\nexcept FileNotFoundError:\n    print(f\"Ошибка: Файл {Config.TRAIN_METADATA_PATH} не найден. Укажите правильный путь.\")\n    # Можно создать dummy DataFrame для продолжения работы над кодом\n    train_df = pd.DataFrame({\n        'filename': [f'bird_{i}.ogg' for i in range(100)],\n        'primary_label': [f'species_{i % 10}' for i in range(100)],\n        'secondary_labels': [[] for _ in range(100)] # Пример пустых вторичных меток\n    })\n    print(\"Создан примерный DataFrame для демонстрации.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-01T00:50:34.473011Z","iopub.execute_input":"2025-05-01T00:50:34.473267Z","iopub.status.idle":"2025-05-01T00:50:34.664147Z","shell.execute_reply.started":"2025-05-01T00:50:34.473245Z","shell.execute_reply":"2025-05-01T00:50:34.663300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Кодирование меток","metadata":{}},{"cell_type":"code","source":"all_labels = set(train_df['primary_label'].unique())\n# Обработка вторичных меток (они могут быть строкой вида '[\"label1\", \"label2\"]')\ndef parse_secondary_labels(labels_str):\n    try:\n        # Используем eval безопасно, т.к. ожидаем формат списка строк\n        labels = eval(labels_str)\n        return [label for label in labels if isinstance(label, str)]\n    except:\n        return []\n\nsecondary_labels_list = train_df['secondary_labels'].apply(parse_secondary_labels).sum()\nall_labels.update(secondary_labels_list)\nsorted_labels = sorted(list(all_labels))\nsorted_labels.remove('')","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Проверяем, соответствует ли количество найденных классов Config.N_CLASSES\nif len(sorted_labels) != Config.N_CLASSES:\n    print(f\"Предупреждение: Найдено {len(sorted_labels)} уникальных меток, но Config.N_CLASSES={Config.N_CLASSES}.\")\n    # Можно обновить Config.N_CLASSES или проверить данные/логику\n    # Config.N_CLASSES = len(sorted_labels) # Раскомментировать для обновления","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Создаем кодировщик","metadata":{}},{"cell_type":"code","source":"label_encoder = LabelEncoder()\nlabel_encoder.fit(sorted_labels)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Преобразование меток в one-hot encoding\ndef get_one_hot_vector(primary_label, secondary_labels_str, encoder):\n    labels = {primary_label}\n    secondary = parse_secondary_labels(secondary_labels_str)\n    labels.update(secondary)\n\n    # Создаем вектор нулей\n    one_hot = np.zeros(len(encoder.classes_), dtype=np.float32)\n\n    # Устанавливаем 1 для присутствующих меток\n    for label in labels:\n        if label in encoder.classes_:\n            idx = encoder.transform([label])[0]\n            one_hot[idx] = 1.0\n    return one_hot\n\ntrain_df['target'] = train_df.apply(\n    lambda row: get_one_hot_vector(row['primary_label'], row['secondary_labels'], label_encoder),\n    axis=1\n)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Разделение на обучающую и валидационную выборки","metadata":{}},{"cell_type":"code","source":"train_indices, val_indices = train_test_split(\n    range(len(train_df)),\n    test_size=0.4, \n    random_state=Config.SEED,\n    # Стратификация может быть полезна, если есть дисбаланс классов\n    stratify=train_df['primary_label'] # Если классов не слишком много\n)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.838Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset и DataLoader","metadata":{}},{"cell_type":"code","source":"def load_and_process_audio(file_path, sr=Config.SAMPLE_RATE, duration=Config.DURATION_SECONDS):\n    try:\n        # Загрузка аудио\n        # Используем res_type='kaiser_fast' для скорости\n        wav, current_sr = librosa.load(file_path, sr=None, res_type='kaiser_fast')\n\n        # Ресемплинг, если необходимо\n        if current_sr != sr:\n            wav = librosa.resample(wav, orig_sr=current_sr, target_sr=sr, res_type='kaiser_fast')\n\n        # Выбор или паддинг до нужной длины\n        target_length = sr * duration\n        if len(wav) > target_length:\n            # Вырезаем случайный сегмент нужной длины\n            max_offset = len(wav) - target_length\n            offset = np.random.randint(max_offset)\n            wav = wav[offset:(offset + target_length)]\n        elif len(wav) < target_length:\n            # Дополняем нулями (паддинг)\n            padding = target_length - len(wav)\n            offset = padding // 2\n            wav = np.pad(wav, (offset, padding - offset), 'constant')\n        else:\n            # Длина совпадает\n            pass\n\n        return wav\n    except Exception as e:\n        print(f\"Ошибка при загрузке {file_path}: {e}\")\n        return np.zeros(sr * duration, dtype=np.float32) # Возвращаем тишину в случае ошибки","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функция для создания мел-спектрограммы\ndef get_mel_spectrogram(wav, sr=Config.SAMPLE_RATE, n_fft=Config.N_FFT,\n                        hop_length=Config.HOP_LENGTH, n_mels=Config.N_MELS,\n                        fmin=Config.FMIN, fmax=Config.FMAX):\n    mel_spec = librosa.feature.melspectrogram(\n        y=wav, sr=sr, n_fft=n_fft, hop_length=hop_length,\n        n_mels=n_mels, fmin=fmin, fmax=fmax\n    )\n    # Преобразование в децибелы (логарифмическая шкала)\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    return mel_spec_db","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" class BirdDataset(Dataset):\n    def __init__(self, df, indices, audio_dir, transforms=None, use_mixup=False, mixup_alpha=0.4):\n        self.df = df.iloc[indices].reset_index(drop=True)\n        self.audio_dir = audio_dir\n        self.transforms = transforms\n        self.use_mixup = use_mixup\n        self.mixup_alpha = mixup_alpha\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        file_path = os.path.join(self.audio_dir, row['filename'])\n\n        # Загрузка и обработка аудио\n        wav = load_and_process_audio(file_path)\n\n        # Получение мел-спектрограммы\n        mel_spec = get_mel_spectrogram(wav)\n\n        # Нормализация спектрограммы (приводим к диапазону [0, 1] или [-1, 1])\n        # Простая нормализация min-max\n        min_val = np.min(mel_spec)\n        max_val = np.max(mel_spec)\n        if max_val > min_val:\n            mel_spec = (mel_spec - min_val) / (max_val - min_val)\n        else:\n             mel_spec = np.zeros_like(mel_spec) # Если все значения одинаковые\n\n        # Преобразование в тензор и добавление канала (как у изображения)\n        image = torch.tensor(mel_spec).unsqueeze(0) # Shape: [1, n_mels, time_steps]\n\n        # Применение трансформаций (аугментаций)\n        if self.transforms:\n            image = self.transforms(image)\n\n        target = torch.tensor(row['target'], dtype=torch.float32)\n\n        # Реализация Mixup [6, 36]\n        if self.use_mixup and random.random() < 0.5: # Применяем Mixup с вероятностью 50%\n            mix_idx = random.randint(0, len(self) - 1)\n            mix_row = self.df.iloc[mix_idx]\n            mix_file_path = os.path.join(self.audio_dir, mix_row['filename'])\n\n            mix_wav = load_and_process_audio(mix_file_path)\n            mix_mel_spec = get_mel_spectrogram(mix_wav)\n            mix_min_val = np.min(mix_mel_spec)\n            mix_max_val = np.max(mix_mel_spec)\n            if mix_max_val > mix_min_val:\n                 mix_mel_spec = (mix_mel_spec - mix_min_val) / (mix_max_val - mix_min_val)\n            else:\n                 mix_mel_spec = np.zeros_like(mix_mel_spec)\n\n            mix_image = torch.tensor(mix_mel_spec).unsqueeze(0)\n            if self.transforms:\n                 mix_image = self.transforms(mix_image)\n\n            mix_target = torch.tensor(mix_row['target'], dtype=torch.float32)\n\n            # Смешивание\n            lam = np.random.beta(self.mixup_alpha, self.mixup_alpha)\n            image = lam * image + (1 - lam) * mix_image\n            target = lam * target + (1 - lam) * mix_target\n\n        return image, target","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Аугментация","metadata":{}},{"cell_type":"code","source":"# SpecAugment [6, 36] - маскирование временных и частотных полос\n# Используем T.SpecAugment из torchvision.transforms v2\ntrain_transforms = T.Compose([\n    T.RandomApply([\n        AT.SpecAugment(\n            # --- Добавленные аргументы ---\n            n_freq_masks=1,  \n            n_time_masks=1,  \n            # --- Существующие аргументы (размер масок) ---\n            freq_mask_param=Config.N_MELS // 8,\n            time_mask_param=int(Config.DURATION_SECONDS * Config.SAMPLE_RATE / Config.HOP_LENGTH / 8)\n        )\n    ], p=0.5) if Config.USE_SPECAUGMENT else nn.Identity(),\n    # ... другие возможные трансформации ...\n])\n\nval_transforms = T.Compose([\n    nn.Identity() # Передаем тождественную(пустую) трансформацию\n])","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Создание датасетов","metadata":{}},{"cell_type":"code","source":"train_dataset = BirdDataset(\n    train_df,\n    train_indices,\n    Config.TRAIN_AUDIO_DIR,\n    transforms=train_transforms,\n    use_mixup=Config.USE_MIXUP,\n    mixup_alpha=Config.MIXUP_ALPHA\n)\nval_dataset = BirdDataset(\n    train_df,\n    val_indices,\n    Config.TRAIN_AUDIO_DIR,\n    transforms=val_transforms, # Без Mixup и SpecAugment для валидации\n    use_mixup=False\n)\n\n# Создание загрузчиков данных\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=Config.BATCH_SIZE,\n    shuffle=True,\n    num_workers=Config.NUM_WORKERS,\n    pin_memory=True # Ускоряет передачу данных на GPU\n)\nval_loader = DataLoader(\n    val_dataset,\n    batch_size=Config.BATCH_SIZE * 2, # Можно увеличить batch_size для валидации\n    shuffle=False,\n    num_workers=Config.NUM_WORKERS,\n    pin_memory=True\n)\n\nprint(f\"Созданы DataLoader: {len(train_loader)} батчей для обучения, {len(val_loader)} для валидации.\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# Загрузка модели EfficientNet с помощью timm [6, 35, 60]\nmodel = timm.create_model(\n    Config.MODEL_NAME,\n    pretrained=Config.PRETRAINED,\n    in_chans=1, # 1 канал для спектрограммы\n    num_classes=Config.N_CLASSES # Количество выходных нейронов\n)\n\ntry:\n    model.load_state_dict(torch.load(Config.MODEL_PATH, map_location=Config.DEVICE))\n    print(f\"Pre-trained weights loaded from {pretrained_weights_path}\")\nexcept FileNotFoundError:\n    print(f\"Error: Pre-trained weights file not found at {pretrained_weights_path}\")\nexcept Exception as e:\n    print(f\"Error loading pre-trained weights: {e}\")\n\n# Перемещение модели на выбранное устройство (GPU или CPU)\nmodel.to(Config.DEVICE)\nprint(f\"Модель {Config.MODEL_NAME} загружена и перемещена на {Config.DEVICE}.\")\n\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Обучение ","metadata":{}},{"cell_type":"code","source":"# Функция потерь и оптимизатор\n# BCEWithLogitsLoss подходит для multi-label классификации (несколько птиц в одном клипе)\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.AdamW(model.parameters(), lr=Config.LEARNING_RATE)\n# Можно добавить планировщик скорости обучения (scheduler)\n# scheduler = optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=Config.EPOCHS)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Цикл обучения\nfor epoch in range(Config.EPOCHS):\n    print(f\"\\n--- Эпоха {epoch+1}/{Config.EPOCHS} ---\")\n\n    # Фаза обучения\n    model.train()\n    train_loss = 0.0\n    train_loop = tqdm(train_loader, desc=\"Обучение\", leave=False)\n    for images, targets in train_loop:\n        images, targets = images.to(Config.DEVICE), targets.to(Config.DEVICE)\n\n        # Прямой проход\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, targets)\n\n        # Обратный проход и оптимизация\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item()\n        train_loop.set_postfix(loss=loss.item())\n\n    avg_train_loss = train_loss / len(train_loader)\n    print(f\"Средняя потеря на обучении: {avg_train_loss:.4f}\")\n\n    # Фаза валидации\n    # model.eval()\n    # val_loss = 0.0\n    # all_preds = []\n    # all_targets = []\n    # val_loop = tqdm(val_loader, desc=\"Валидация\", leave=False)\n    # with torch.no_grad(): # Отключаем вычисление градиентов\n    #     for images, targets in val_loop:\n    #         images, targets = images.to(Config.DEVICE), targets.to(Config.DEVICE)\n\n    #         # Прямой проход\n    #         outputs = model(images)\n    #         loss = criterion(outputs, targets)\n    #         val_loss += loss.item()\n\n    #         # Сохраняем предсказания (вероятности после сигмоиды) и цели\n    #         preds = torch.sigmoid(outputs)\n    #         all_preds.append(preds.cpu().numpy())\n    #         all_targets.append(targets.cpu().numpy())\n\n    # avg_val_loss = val_loss / len(val_loader)\n    # print(f\"Средняя потеря на валидации: {avg_val_loss:.4f}\")\n\n    # # Расчет метрики ROC-AUC (требует sklearn)\n    # try:\n    #     from sklearn.metrics import roc_auc_score\n    #     all_preds = np.concatenate(all_preds)\n    #     all_targets = np.concatenate(all_targets)\n\n    #     # Расчет макро-усредненной ROC-AUC [27]\n    #     # Обработка случая, когда в батче нет примеров какого-то класса\n    #     valid_targets = all_targets.sum(axis=0) > 0\n    #     macro_roc_auc = roc_auc_score(\n    #         all_targets[:, valid_targets], # Берем только классы, которые были в валидации\n    #         all_preds[:, valid_targets],\n    #         average='macro'\n    #     )\n    #     print(f\"Макро ROC-AUC на валидации: {macro_roc_auc:.4f}\")\n    # except ImportError:\n    #     print(\"Библиотека scikit-learn не найдена. Пропустите расчет ROC-AUC.\")\n    # except Exception as e:\n    #     print(f\"Ошибка при расчете ROC-AUC: {e}\")\n\n\n    # Обновление планировщика (если используется)\n    # scheduler.step()\n\n    # TODO: Сохранение лучшей модели на основе валидационной метрики (например, ROC-AUC)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.840Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Инференс","metadata":{}},{"cell_type":"code","source":"def load_audio_segment(file_path, start_time, duration, sr=Config.SAMPLE_RATE):\n    \"\"\"Loads a specific segment from an audio file.\"\"\"\n    try:\n        # Load the audio, potentially just the required segment\n        # Using offset and duration in librosa load\n        # Ensure duration doesn't exceed file length from offset\n        info = torchaudio.info(str(file_path))\n        file_duration = info.num_frames / info.sample_rate\n        actual_duration_to_load = min(duration, file_duration - start_time)\n        if actual_duration_to_load <= 0:\n             return np.zeros(int(duration * sr), dtype=np.float32) # Segment is beyond file end\n\n        wav, current_sr = librosa.load(\n            file_path,\n            sr=None, # Load at original sample rate\n            offset=start_time, # Start loading from this time\n            duration=actual_duration_to_load, # Load for this duration\n            res_type='kaiser_fast'\n        )\n\n        # Resample if necessary\n        if current_sr != sr:\n            wav = librosa.resample(wav, orig_sr=current_sr, target_sr=sr, res_type='kaiser_fast')\n\n        # Pad if the loaded segment is shorter than expected (e.g., at the very end of the file)\n        target_length = int(duration * sr)\n        if len(wav) < target_length:\n            padding = target_length - len(wav)\n            wav = np.pad(wav, (0, padding), 'constant')\n        elif len(wav) > target_length:\n             # Should not happen with offset+duration unless calculation is off\n             wav = wav[:target_length]\n\n        return wav\n\n    except Exception as e:\n        print(f\"Error loading segment from {file_path} at {start_time}s: {e}\")\n        return np.zeros(int(duration * sr), dtype=np.float32) # Return silence on error\n\n\ndef normalize_spectrogram(mel_spec):\n    \"\"\"Normalizes mel spectrogram similar to training.\"\"\"\n    min_val = np.min(mel_spec)\n    max_val = np.max(mel_spec)\n    if max_val > min_val:\n        # Simple min-max scaling to [0, 1]\n        mel_spec = (mel_spec - min_val) / (max_val - min_val)\n    else:\n         # Handle case with constant values (e.g., silence)\n         mel_spec = np.zeros_like(mel_spec)\n    return mel_spec\n\n\n# --- Test Dataset ---\nclass TestSoundscapesDataset(Dataset):\n    def __init__(self, segment_list, audio_dir, config):\n        \"\"\"\n        Args:\n            segment_list (list): List of tuples (filename, start_time_seconds, soundscape_id).\n                                 soundscape_id is extracted from filename (e.g., '12345' from 'soundscape_12345.ogg')\n            audio_dir (str): Directory containing the soundscape audio files.\n            config: Configuration object.\n        \"\"\"\n        self.segment_list = segment_list\n        self.audio_dir = audio_dir\n        self.config = config\n\n    def __len__(self):\n        return len(self.segment_list)\n\n    def __getitem__(self, idx):\n        filename, start_time, soundscape_id = self.segment_list[idx]\n        file_path = os.path.join(self.audio_dir, filename)\n\n        # Load the specific segment\n        wav = load_audio_segment(file_path, start_time, self.config.DURATION_SECONDS, self.config.SAMPLE_RATE)\n\n        # Get mel spectrogram\n        mel_spec = get_mel_spectrogram(\n            wav,\n            sr=self.config.SAMPLE_RATE,\n            n_fft=self.config.N_FFT,\n            hop_length=self.config.HOP_LENGTH,\n            n_mels=self.config.N_MELS,\n            fmin=self.config.FMIN,\n            fmax=self.config.FMAX\n        )\n\n        # Normalize\n        mel_spec = normalize_spectrogram(mel_spec)\n\n        # Convert to tensor and add channel dimension\n        # Expected shape by EfficientNet is [Batch, Channels, Height, Width] -> [B, 1, N_Mels, TimeSteps]\n        image = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0)\n\n        # Create row_id: soundscape_[soundscape_id]_[end_time]\n        end_time = int(start_time + self.config.DURATION_SECONDS) # End time is start + duration\n        row_id = f\"soundscape_{soundscape_id}_{end_time}\"\n\n        return image, row_id\n\n# --- Main Submission Logic ---\nprint(\"Starting submission script...\")\n\n\n# --- Load Sample Submission to get required row_ids and species columns ---\nsample_submission_df = pd.read_csv(Config.SAMPLE_SUBMISSION_PATH)\nrequired_row_ids = sample_submission_df['row_id'].tolist()\nsubmission_species_cols = sample_submission_df.columns[1:].tolist()\nprint(f\"Loaded {len(required_row_ids)} row IDs and {len(submission_species_cols)} species columns from sample submission.\")\n\nif len(submission_species_cols) != Config.N_CLASSES:\n     print(f\"FATAL Error: Number of species columns in sample submission ({len(submission_species_cols)}) does not match Config.N_CLASSES ({Config.N_CLASSES}). Check configuration or data.\")\n\n\n# --- Load Label Encoder ---\n# We need the label encoder that maps species names to indices (0 to N_CLASSES-1)\n# based on how your model was trained. It's crucial this mapping is consistent.\n# The safest way is to rebuild it using the same sorted list of ALL species\n# found in the training data primary/secondary labels.\nprint(\"Creating/Loading label encoder...\")\ntry:\n    train_df_for_labels = pd.read_csv(Config.TRAIN_METADATA_PATH)\n    all_labels_set = set(train_df_for_labels['primary_label'].unique())\n    def parse_secondary_labels_safe(labels_str):\n        try:\n            # Use ast.literal_eval for safer evaluation if the string is a list literal\n            import ast\n            labels = ast.literal_eval(labels_str)\n            return [label for label in labels if isinstance(label, str)]\n        except:\n            return [] # Return empty list on error\n\n    secondary_labels_list_for_labels = train_df_for_labels['secondary_labels'].apply(parse_secondary_labels_safe).sum()\n    all_labels_set.update(secondary_labels_list_for_labels)\n    sorted_labels_for_encoder = sorted(list(all_labels_set))\n    if '' in sorted_labels_for_encoder: # Remove potential empty string if present\n        sorted_labels_for_encoder.remove('')\n\n    label_encoder = LabelEncoder()\n    label_encoder.fit(sorted_labels_for_encoder)\n\n    print(f\"Label encoder fitted with {len(label_encoder.classes_)} classes based on training data.\")\n    if len(label_encoder.classes_) != Config.N_CLASSES:\n         print(f\"Warning: Number of classes in encoder ({len(label_encoder.classes_)}) does not match Config.N_CLASSES ({Config.N_CLASSES}). Check your training data / Config.N_CLASSES.\")\n         # You might need to adjust N_CLASSES in Config if the actual number of species in train.csv is different\n         # Or check why the species list derived from training data doesn't match 206.\n\nexcept FileNotFoundError as e:\n    print(f\"Error loading training data for label encoder: {e}. Cannot build robust encoder.\")\n    # If you cannot load train.csv, you might have issues mapping model outputs correctly.\n    # This is a critical point - the mapping MUST be correct.\n    # A potential fallback is to use the species names from sample_submission and hope they cover everything,\n    # but this is risky if your model was trained on more/different species.\n    print(\"FATAL Error: Could not build label encoder from training data. Please check paths.\")\nexcept Exception as e:\n    print(f\"An unexpected error occurred while building label encoder: {e}\")\n\n# --- Create a mapping from species ID string (from submission columns) to encoder index ---\n# This maps the required output column to the index in the model's output tensor\nspecies_id_to_encoder_idx = {\n    species_id: label_encoder.transform([species_id])[0]\n    for species_id in submission_species_cols\n    if species_id in label_encoder.classes_ # Ensure species from submission are known to encoder\n}\n# Check if all submission species are known\nif len(species_id_to_encoder_idx) != len(submission_species_cols):\n     missing_species = [sp for sp in submission_species_cols if sp not in label_encoder.classes_]\n     print(f\"Warning: The following species from sample_submission are NOT in the training data's label encoder: {missing_species}\")\n     # Predictions for these species will likely be zeros or based on random chance, depending on model's final layer initialization\n\n# --- Load Model ---\n# print(f\"Loading model from {Config.SAVED_MODEL_PATH}...\")\n# model = timm.create_model(\n#     Config.MODEL_NAME,\n#     pretrained=False, # We are loading custom weights\n#     in_chans=1,\n#     num_classes=Config.N_CLASSES # Model output size must match the number of classes the encoder knows\n# )\n\n# try:\n#     model.load_state_dict(torch.load(Config.SAVED_MODEL_PATH, map_location=Config.DEVICE))\n#     print(\"Model weights loaded successfully.\")\n# except FileNotFoundError:\n#     print(f\"FATAL Error: Saved model not found at {Config.SAVED_MODEL_PATH}. Please update Config.SAVED_MODEL_PATH.\")\n#     exit() # Exit if model weights are not found\n# except Exception as e:\n#     print(f\"Error loading model state dict: {e}\")\n#     exit() # Exit on other loading errors\n\n# model.to(Config.DEVICE)\n# model.eval() # Set model to evaluation mode\n\n# --- Prepare Test Data Segments ---\nprint(\"Preparing test data segments...\")\ntest_audio_files = list(pathlib.Path(Config.TEST_SOUNDSCAPES_DIR).glob(\"*.ogg\"))\nall_test_segments = []\n\nfor file_path in tqdm(test_audio_files, desc=\"Calculating test segments\"):\n    filename = file_path.name\n    soundscape_id = pathlib.Path(filename).stem.split('_')[-1] # Extract ID from soundscape_xxxxxx.ogg\n    try:\n        info = torchaudio.info(str(file_path))\n        file_duration = info.num_frames / info.sample_rate\n\n        # Calculate segment start times (0, 5, 10, ..., 55 for a 60s file)\n        # The last segment might be shorter if the file duration is not a multiple of DURATION_SECONDS\n        # Kaggle usually evaluates 5s segments starting at 0, 5, 10...\n        segment_starts = np.arange(0, file_duration, Config.DURATION_SECONDS)\n\n        for start_time in segment_starts:\n            # Ensure we don't create segments starting exactly at the end of the file\n            if start_time < file_duration:\n                all_test_segments.append((filename, start_time, soundscape_id))\n\n    except Exception as e:\n        print(f\"Could not process {filename} for segmentation: {e}\")\n        # Optionally skip this file or add dummy segments if needed\n\nprint(f\"Generated {len(all_test_segments)} segments for inference.\")\n\n# Create Test Dataset and DataLoader\ntest_dataset = TestSoundscapesDataset(all_test_segments, Config.TEST_SOUNDSCAPES_DIR, Config)\ntest_loader = DataLoader(\n    test_dataset,\n    batch_size=Config.INFERENCE_BATCH_SIZE,\n    shuffle=False, # Maintain order\n    num_workers=Config.NUM_WORKERS,\n    pin_memory=True\n)\n\nprint(f\"Created Test DataLoader with {len(test_loader)} batches.\")\n\n# --- Run Inference ---\nprint(\"Running inference...\")\n# Lists to store results before creating DataFrame\nall_row_ids = []\nall_probability_vectors = [] # Store the full probability vector for each segment\n\nwith torch.no_grad(): # Disable gradient calculations for inference\n    test_loop = tqdm(test_loader, desc=\"Inference\", leave=True) # Set leave=True to show final bar\n    for images, batch_row_ids in test_loop:\n        images = images.to(Config.DEVICE)\n\n        # Get predictions (logits)\n        outputs = model(images)\n\n        # Apply sigmoid to get probabilities (0-1)\n        probabilities = torch.sigmoid(outputs) # Shape: [batch_size, N_CLASSES]\n\n        # Store row IDs and probability vectors\n        all_row_ids.extend(batch_row_ids)\n        all_probability_vectors.append(probabilities.cpu().numpy())\n\nprint(\"Inference complete.\")\n\n# Concatenate all probability vectors from batches\nif all_probability_vectors:\n    all_probability_vectors = np.concatenate(all_probability_vectors, axis=0)\n    print(f\"Combined probability vectors shape: {all_probability_vectors.shape}\")\nelse:\n    print(\"No segments were processed. Probability matrix is empty.\")\n    all_probability_vectors = np.empty((0, Config.N_CLASSES)) # Handle empty case\n\n# --- Create Submission DataFrame ---\nprint(f\"Creating submission file: {Config.SUBMISSION_CSV_PATH}\")\n\n# Initialize the submission DataFrame with required columns\nsubmission_df = pd.DataFrame(index=range(len(all_row_ids)), columns=['row_id'] + submission_species_cols)\n\n# Populate the row_id column\nsubmission_df['row_id'] = all_row_ids\n\n# Populate the species probability columns\nfor species_id in submission_species_cols:\n    if species_id in species_id_to_encoder_idx:\n        # Get the index this species corresponds to in the model's output tensor\n        encoder_idx = species_id_to_encoder_idx[species_id]\n        # Assign the probabilities from the collected vectors\n        submission_df[species_id] = all_probability_vectors[:, encoder_idx]\n    else:\n        # If a species from sample submission wasn't in the training encoder, predict 0 probability\n        print(f\"Warning: Filling column '{species_id}' with zeros as it was not found in the training label encoder.\")\n        submission_df[species_id] = 0.0 # Or 0.5, or some other default\n\n# --- Ensure all required row_ids from sample_submission are present ---\n# This is a critical step to avoid submission errors.\n# Create a DataFrame from our results\nour_results_df = submission_df.copy()\n\n# Merge our results with the sample submission row_ids\n# This keeps all rows from sample_submission and fills in probabilities where we have them\nfinal_submission_df = sample_submission_df[['row_id']].merge(our_results_df, on='row_id', how='left')\n\n# Fill any row_ids that were in sample_submission but not in our generated segments\n# This might happen if there was an error processing a file, or if the segmentation logic slightly differs.\n# Fill probability columns with 0.0 for missing rows/species\n# The merge will add columns from our_results_df. We only need to fill the species columns.\nfor col in submission_species_cols:\n     if col in final_submission_df.columns:\n        final_submission_df[col] = final_submission_df[col].fillna(0.0) # Fill with 0.0 probability\n     else:\n        # This should not happen if our_results_df was created correctly based on sample_submission_cols\n        final_submission_df[col] = 0.0\n        print(f\"Error: Column '{col}' missing in merged DataFrame. Adding with zeros.\")\n\n# The 'row_id' column from the merge should not have NaNs if sample_submission was loaded correctly.\n\nprint(f\"Final submission DataFrame shape: {final_submission_df.shape}\")\nprint(f\"Final submission columns: {final_submission_df.columns.tolist()}\")\n\n# Save the submission file\nfinal_submission_df.to_csv(Config.SUBMISSION_CSV_PATH, index=False)\n\nprint(\"Submission file created successfully!\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-05-01T01:32:30.841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}