{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":14168060,"sourceType":"datasetVersion","datasetId":9031034}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport csv\nimport random\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold\nfrom torchvision.models import resnet50\nfrom torch.utils.data import Dataset, DataLoader\nfrom skimage import exposure, util\nfrom skimage.transform import resize\nfrom skimage.filters import gaussian\nimport librosa\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Используемое устройство: {DEVICE}\")\n\n# Константы\nNUM_CLASSES = 24\nSAMPLE_RATE = 48000\nCLIP_DURATION_SEC = 10\nCLIP_LENGTH_SAMPLES = CLIP_DURATION_SEC * SAMPLE_RATE\nINITIAL_FMIN = float(\"inf\")\nINITIAL_FMAX = float(\"-inf\")\nLEARNING_RATE = 2e-4\nEPOCHS = 20\nNUM_FOLDS = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:55:55.384494Z","iopub.execute_input":"2025-12-15T18:55:55.384780Z","iopub.status.idle":"2025-12-15T18:55:59.547746Z","shell.execute_reply.started":"2025-12-15T18:55:55.384760Z","shell.execute_reply":"2025-12-15T18:55:59.546987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функции обработки изображений\n\nclass AudioAugmentationPipeline:\n    def __init__(self):\n        self.transforms = [\n            self._horizontal_flip,\n            self._vertical_flip,\n            self._add_gaussian_noise,\n            self._rescale_contrast\n        ]\n\n    def _horizontal_flip(self, img):\n        return np.stack([img[:, ::-1]] * 3, axis=0)\n\n    def _vertical_flip(self, img):\n        return np.stack([img[::-1, :]] * 3, axis=0)\n\n    def _add_gaussian_noise(self, img):\n        noisy = util.random_noise(img)\n        return np.stack([noisy] * 3, axis=0)\n\n    def _rescale_contrast(self, img):\n        enhanced = exposure.rescale_intensity(img)\n        return np.stack([enhanced] * 3, axis=0)\n\n    def apply_random_transform(self, img):\n        transform = random.choice(self.transforms)\n        return transform(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:55:59.548963Z","iopub.execute_input":"2025-12-15T18:55:59.549333Z","iopub.status.idle":"2025-12-15T18:55:59.556614Z","shell.execute_reply.started":"2025-12-15T18:55:59.549303Z","shell.execute_reply":"2025-12-15T18:55:59.555819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Преобразование спектрограммы в нормализованное изображение\n\ndef spectrogram_to_image(spec: np.ndarray) -> np.ndarray:\n    resized = resize(spec, (224, 400), anti_aliasing=True)\n    eps = 1e-6\n\n    normalized = (resized - resized.mean()) / (resized.std() + eps)\n    min_val, max_val = normalized.min(), normalized.max()\n    scaled = 255 * (normalized - min_val) / (max_val - min_val + eps)\n    return scaled.astype(np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:55:59.557435Z","iopub.execute_input":"2025-12-15T18:55:59.557778Z","iopub.status.idle":"2025-12-15T18:55:59.571317Z","shell.execute_reply.started":"2025-12-15T18:55:59.557756Z","shell.execute_reply":"2025-12-15T18:55:59.570652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Создание модели\ndef build_model(num_classes=24):\n    model = resnet50(weights=None)\n    weights_path = \"/kaggle/input/resnet50-weights-offline/resnet50-weights.pth\"\n    model.load_state_dict(torch.load(weights_path, map_location=DEVICE))\n    model.fc = nn.Linear(model.fc.in_features, num_classes)\n    return model.to(DEVICE)\n    \n# Загрузка и анализ метаданных\n\ntrain_metadata = pd.read_csv(\"/kaggle/input/rfcx-species-audio-detection/train_tp.csv\")\n\nf_min = min(train_metadata[\"f_min\"]) * 0.9\nf_max = max(train_metadata[\"f_max\"]) * 1.1\nF_MIN = int(f_min)\nF_MAX = int(f_max)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:55:59.572028Z","iopub.execute_input":"2025-12-15T18:55:59.572640Z","iopub.status.idle":"2025-12-15T18:55:59.600026Z","shell.execute_reply.started":"2025-12-15T18:55:59.572589Z","shell.execute_reply":"2025-12-15T18:55:59.599307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обработка аудиофайлов\n\nrecording_ids = train_metadata[\"recording_id\"].tolist()\nspecies_labels = train_metadata[\"species_id\"].tolist()\nprecomputed_spectrograms = {}\n\ndef extract_spectrogram(index: int):\n    rec_id = recording_ids[index]\n    label = species_labels[index]\n\n    filepath = f\"/kaggle/input/rfcx-species-audio-detection/train/{rec_id}.flac\"\n    audio, sr = librosa.load(filepath, sr=None)\n\n    t_min_sec = train_metadata.at[index, \"t_min\"]\n    t_max_sec = train_metadata.at[index, \"t_max\"]\n    t_min = int(t_min_sec * sr)\n    t_max = int(t_max_sec * sr)\n\n    center = (t_min + t_max) // 2\n    start = max(center - CLIP_LENGTH_SAMPLES // 2, 0)\n    end = min(start + CLIP_LENGTH_SAMPLES, len(audio))\n    if end - start < CLIP_LENGTH_SAMPLES:\n        start = end - CLIP_LENGTH_SAMPLES\n\n    segment = audio[int(start):int(end)]\n    mel_spec = librosa.feature.melspectrogram(y=segment, sr=sr, fmin=F_MIN, fmax=F_MAX)\n    db_spec = librosa.power_to_db(mel_spec, top_db=80)\n\n    image = spectrogram_to_image(db_spec)\n    return rec_id, image\n\n# Параллельная предварительная обработка\nwith ThreadPoolExecutor() as executor:\n    results = list(tqdm(executor.map(extract_spectrogram, range(len(recording_ids))), \n                        total=len(recording_ids), desc=\"Предварительная обработка\"))\n    precomputed_spectrograms.update(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:55:59.602175Z","iopub.execute_input":"2025-12-15T18:55:59.602449Z","iopub.status.idle":"2025-12-15T18:56:59.118008Z","shell.execute_reply.started":"2025-12-15T18:55:59.602427Z","shell.execute_reply":"2025-12-15T18:56:59.117254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Датасет для обучения\n\nclass SpectrogramDataset(Dataset):\n    def __init__(self, recording_ids, labels, split_type, augmentation_pipeline=None):\n        self.recording_ids = recording_ids\n        self.labels = labels\n        self.split_type = split_type\n        self.augmentation = augmentation_pipeline\n        self.cache = precomputed_spectrograms\n\n    def __len__(self):\n        return len(self.recording_ids)\n\n    def __getitem__(self, idx):\n        rec_id = self.recording_ids[idx]\n        label = self.labels[idx]\n        spec_img = self.cache[rec_id]\n\n        if self.split_type == \"train\" and self.augmentation:\n            image = self.augmentation.apply_random_transform(spec_img)\n        else:\n            image = np.stack([spec_img] * 3, axis=0)\n\n        return torch.from_numpy(image).float(), torch.tensor(label, dtype=torch.long)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:56:59.118814Z","iopub.execute_input":"2025-12-15T18:56:59.119232Z","iopub.status.idle":"2025-12-15T18:56:59.126653Z","shell.execute_reply.started":"2025-12-15T18:56:59.119211Z","shell.execute_reply":"2025-12-15T18:56:59.125587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обучение модели с валидацией\n\ndef train_model(model, criterion, train_loader, val_loader, optimizer, scheduler):\n    best_acc = 0.0\n    best_weights = None\n\n    for epoch in range(1, EPOCHS + 1):\n        model.train()\n        train_loss = []\n\n        for inputs, targets in train_loader:\n            inputs = inputs.to(DEVICE)\n            targets = targets.to(DEVICE)\n\n            optimizer.zero_grad()\n            outputs = model(inputs)\n            loss = criterion(outputs, targets)\n            loss.backward()\n            optimizer.step()\n            train_loss.append(loss.item())\n\n        model.eval()\n        val_loss = []\n        all_preds = []\n        all_labels = []\n\n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs = inputs.to(DEVICE)\n                targets = targets.to(DEVICE)\n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n                val_loss.append(loss.item())\n                all_preds.append(outputs.cpu().numpy())\n                all_labels.append(targets.cpu().numpy())\n\n        all_preds = np.concatenate(all_preds, axis=0)\n        all_labels = np.concatenate(all_labels, axis=0)\n        accuracy = np.mean(np.argmax(all_preds, axis=1) == all_labels)\n\n        avg_train_loss = np.mean(train_loss)\n        avg_val_loss = np.mean(val_loss)\n        print(f\"Эпоха {epoch:02d} | Train loss: {avg_train_loss:.5f} | \"\n              f\"Val loss: {avg_val_loss:.5f} | Val acc: {accuracy:.5f}\")\n\n        scheduler.step(avg_val_loss)\n        if accuracy > best_acc:\n            best_acc = accuracy\n            best_weights = model.state_dict()\n\n    model.load_state_dict(best_weights)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:56:59.127543Z","iopub.execute_input":"2025-12-15T18:56:59.127741Z","iopub.status.idle":"2025-12-15T18:56:59.144116Z","shell.execute_reply.started":"2025-12-15T18:56:59.127724Z","shell.execute_reply":"2025-12-15T18:56:59.143374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Кросс-валидация и обучение\n\nkfold = KFold(n_splits=NUM_FOLDS, shuffle=True, random_state=563)\naugmenter = AudioAugmentationPipeline()\n\nfor fold, (train_idx, val_idx) in enumerate(kfold.split(recording_ids)):\n    print(f\"\\nОбучение на фолде {fold}\")\n\n    X_train = [recording_ids[i] for i in train_idx]\n    y_train = [species_labels[i] for i in train_idx]\n    X_val = [recording_ids[i] for i in val_idx]\n    y_val = [species_labels[i] for i in val_idx]\n\n    train_dataset = SpectrogramDataset(X_train, y_train, \"train\", augmentation_pipeline=augmenter)\n    val_dataset = SpectrogramDataset(X_val, y_val, \"valid\")\n\n    train_loader = DataLoader(train_dataset, batch_size=8, shuffle=True, drop_last=True)\n    val_loader = DataLoader(val_dataset, batch_size=8, shuffle=False, drop_last=False)\n\n    model = build_model()\n    optimizer = torch.optim.Adam(model.parameters(), lr=LEARNING_RATE)\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode=\"min\", patience=3)\n\n    trained_model = train_model(model, nn.CrossEntropyLoss(), train_loader, val_loader, optimizer, scheduler)\n    torch.save(trained_model.state_dict(), f\"./model{fold}.pt\")\n\n    del train_dataset, val_dataset, train_loader, val_loader, trained_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:56:59.145083Z","iopub.execute_input":"2025-12-15T18:56:59.145350Z","iopub.status.idle":"2025-12-15T18:57:33.462290Z","shell.execute_reply.started":"2025-12-15T18:56:59.145322Z","shell.execute_reply":"2025-12-15T18:57:33.460226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Загрузка ансамбля моделей\n\nensemble_models = []\nfor i in range(NUM_FOLDS):\n    model = build_model()\n    model.load_state_dict(torch.load(f\"./model{i}.pt\", map_location=DEVICE))\n    model.eval()\n    ensemble_models.append(model)\n    os.remove(f\"./model{i}.pt\")\n\nif torch.cuda.is_available():\n    ensemble_models = [m.cuda() for m in ensemble_models]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:57:33.463260Z","iopub.status.idle":"2025-12-15T18:57:33.463616Z","shell.execute_reply.started":"2025-12-15T18:57:33.463461Z","shell.execute_reply":"2025-12-15T18:57:33.463477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обработка тестовых данных\n\ndef process_test_file(filename: str):\n    path = f\"/kaggle/input/rfcx-species-audio-detection/test/{filename}\"\n    audio, sr = librosa.load(path, sr=None)\n\n    num_segments = int(np.ceil(len(audio) / CLIP_LENGTH_SAMPLES))\n    segments = []\n\n    for i in range(num_segments):\n        start = i * CLIP_LENGTH_SAMPLES\n        end = start + CLIP_LENGTH_SAMPLES\n        if end > len(audio):\n            segment = audio[-CLIP_LENGTH_SAMPLES:]\n        else:\n            segment = audio[start:end]\n\n        mel = librosa.feature.melspectrogram(y=segment, sr=sr, fmin=F_MIN, fmax=F_MAX)\n        db_mel = librosa.power_to_db(mel, top_db=80)\n        img = spectrogram_to_image(db_mel)\n        rgb_img = np.stack([img] * 3, axis=0)\n        segments.append(rgb_img)\n\n    return torch.tensor(np.stack(segments, axis=0), dtype=torch.float32)\n\ndef predict_on_file(filename: str, models):\n    file_id = filename.split(\".\")[0]\n    segments = process_test_file(filename)\n\n    if torch.cuda.is_available():\n        segments = segments.cuda()\n\n    predictions = []\n    for model in models:\n        with torch.no_grad():\n            outputs = model(segments)\n            max_per_class = torch.max(outputs, dim=0).values\n            predictions.append(max_per_class.cpu())\n\n    averaged = torch.mean(torch.stack(predictions), dim=0)\n    return [file_id] + [val.item() for val in averaged]\n\ndef create_submission(test_files, models, output_path=\"submission.csv\"):\n    header = [\"recording_id\"] + [f\"s{i}\" for i in range(NUM_CLASSES)]\n    with open(output_path, \"w\", newline=\"\", encoding=\"utf-8\") as f:\n        writer = csv.writer(f)\n        writer.writerow(header)\n        with ThreadPoolExecutor(max_workers=4) as executor:\n            futures = [executor.submit(predict_on_file, fn, models) for fn in test_files]\n            for future in tqdm(futures, desc=\"Создание предсказаний\"):\n                writer.writerow(future.result())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:57:33.464869Z","iopub.status.idle":"2025-12-15T18:57:33.465140Z","shell.execute_reply.started":"2025-12-15T18:57:33.465014Z","shell.execute_reply":"2025-12-15T18:57:33.465025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Запуск на тестовых данных\n\ntest_files = os.listdir(\"/kaggle/input/rfcx-species-audio-detection/test/\")\nprint(f\"Найдено тестовых файлов: {len(test_files)}\")\n\ncreate_submission(test_files, ensemble_models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T18:57:33.466473Z","iopub.status.idle":"2025-12-15T18:57:33.466798Z","shell.execute_reply.started":"2025-12-15T18:57:33.466666Z","shell.execute_reply":"2025-12-15T18:57:33.466679Z"}},"outputs":[],"execution_count":null}]}