{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Лабораторная работа №3 \n# Rainforest Connection Species Audio Detection\n\nВыполнил ***Орлов А.Л.*** студент группы ***0307*** ","metadata":{}},{"cell_type":"markdown","source":"## Импорты и конфигурация","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport librosa\nfrom tqdm import tqdm\nfrom skimage.transform import resize\nfrom PIL import Image\nimport pandas as pd\nimport warnings\nimport random\nimport torch\nimport torch.utils.data as torchdata\nfrom sklearn.model_selection import StratifiedKFold\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport timm\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n\nfft = 2048\nhop = 512\nsr = 48000\nlength = 10 * sr\nWORKING_DIR = '/kaggle/working/'\nAUDIO_DATA = '/kaggle/input/rfcx-species-audio-detection/train/'\nTRAIN_TP = '/kaggle/input/rfcx-species-audio-detection/train_tp.csv'\ndf = pd.read_csv(TRAIN_TP)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:20:13.738027Z","iopub.execute_input":"2025-12-05T02:20:13.738349Z","iopub.status.idle":"2025-12-05T02:20:13.747946Z","shell.execute_reply.started":"2025-12-05T02:20:13.738327Z","shell.execute_reply":"2025-12-05T02:20:13.747350Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Загрузка спектограмм","metadata":{}},{"cell_type":"code","source":"# Определяем диапазон частот\nfmin = int(df['f_min'].min() * 0.9)\nfmax = int(df['f_max'].max() * 1.1)\n\nfor idx, row in tqdm(df.iterrows(), total=len(df), desc='Получение спектрограмм'):\n    wav, sr = librosa.load(f\"{AUDIO_DATA}{row['recording_id']}.flac\", sr=None)\n    \n    t_min = float(row['t_min']) * sr\n    t_max = float(row['t_max']) * sr\n    \n    center = np.round((t_min + t_max) / 2)\n    beginning = center - length / 2\n    if beginning < 0:\n        beginning = 0\n    \n    ending = beginning + length\n    if ending > len(wav):\n        ending = len(wav)\n        beginning = ending - length\n        \n    slice = wav[int(beginning):int(ending)]\n    \n    mel_spec = librosa.feature.melspectrogram(\n        y=slice, n_fft=fft, hop_length=hop, sr=sr, fmin=fmin, fmax=fmax, power=1.5\n    )\n    mel_spec = resize(mel_spec, (224, 400))\n    \n    mel_spec = mel_spec - np.min(mel_spec)\n    mel_spec = mel_spec / np.max(mel_spec)\n    mel_spec = (mel_spec*255).astype('uint8')\n    \n    bmp = Image.fromarray(mel_spec, 'L')\n    bmp.save(f\"{WORKING_DIR}{row['recording_id']}_{row['species_id']}_{int(center)}.bmp\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:15:17.702142Z","iopub.execute_input":"2025-12-05T02:15:17.702886Z","iopub.status.idle":"2025-12-05T02:17:49.031516Z","shell.execute_reply.started":"2025-12-05T02:15:17.702859Z","shell.execute_reply":"2025-12-05T02:17:49.030672Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset и DataLoader","metadata":{}},{"cell_type":"code","source":"class RainforestDataset(Dataset):\n    def __init__(self, file_list, working_dir=WORKING_DIR, num_classes=24):\n        self.specs = []\n        self.labels = []\n        self.num_classes = num_classes\n\n        for f in file_list:\n            # Метка\n            label = int(f.split('_')[1])\n            label_array = np.zeros(self.num_classes, dtype=np.float32)\n            label_array[label] = 1.\n            self.labels.append(label_array)\n\n            # Спектрограмма\n            img = Image.open(os.path.join(working_dir, f))\n            mel_spec = np.array(img).astype(np.float32) / 255.\n            img.close()\n\n            # 3 канала\n            mel_spec = np.stack([mel_spec, mel_spec, mel_spec], axis=0)\n            self.specs.append(mel_spec)\n\n    def __len__(self):\n        return len(self.specs)\n\n    def __getitem__(self, idx):\n        return torch.tensor(self.specs[idx], dtype=torch.float32), torch.tensor(self.labels[idx], dtype=torch.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:20:18.259596Z","iopub.execute_input":"2025-12-05T02:20:18.259885Z","iopub.status.idle":"2025-12-05T02:20:18.266122Z","shell.execute_reply.started":"2025-12-05T02:20:18.259861Z","shell.execute_reply":"2025-12-05T02:20:18.265520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_list = [f for f in os.listdir(WORKING_DIR) if f.endswith('.bmp')]\nlabel_list = [int(f.split('_')[1]) for f in file_list]\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=1234)\nfor fold_id, (train_idx, val_idx) in enumerate(skf.split(file_list, label_list)):\n    if fold_id == 0:\n        train_files = np.take(file_list, train_idx)\n        val_files = np.take(file_list, val_idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:20:21.632049Z","iopub.execute_input":"2025-12-05T02:20:21.632780Z","iopub.status.idle":"2025-12-05T02:20:21.641326Z","shell.execute_reply.started":"2025-12-05T02:20:21.632753Z","shell.execute_reply":"2025-12-05T02:20:21.640599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 16\n\ntrain_dataset = RainforestDataset(train_files)\nval_dataset = RainforestDataset(val_files)\n\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:20:23.725140Z","iopub.execute_input":"2025-12-05T02:20:23.725764Z","iopub.status.idle":"2025-12-05T02:20:25.129210Z","shell.execute_reply.started":"2025-12-05T02:20:23.725735Z","shell.execute_reply":"2025-12-05T02:20:25.128659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Модель","metadata":{}},{"cell_type":"code","source":"num_birds = 24\nmodel = timm.create_model('resnest101e', pretrained=True)\n\nmodel.fc = nn.Sequential(\n    nn.Linear(2048, 1024),\n    nn.ReLU(),\n    nn.Dropout(0.2),\n    nn.Linear(1024, 1024),\n    nn.ReLU(),\n    nn.Dropout(0.2),\n    nn.Linear(1024, num_birds)\n)\n\nif torch.cuda.is_available():\n    model = model.cuda()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:20:27.781689Z","iopub.execute_input":"2025-12-05T02:20:27.782393Z","iopub.status.idle":"2025-12-05T02:20:31.132926Z","shell.execute_reply.started":"2025-12-05T02:20:27.782368Z","shell.execute_reply":"2025-12-05T02:20:31.132047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Обучающий цикл (с валидацией)","metadata":{}},{"cell_type":"code","source":"optimizer = torch.optim.SGD(model.parameters(), lr=0.01, momentum=0.9, weight_decay=1e-4)\nscheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=7, gamma=0.4)\nloss_fn = nn.BCEWithLogitsLoss(pos_weight=torch.ones(num_birds) * num_birds)\nif torch.cuda.is_available():\n    loss_fn = loss_fn.cuda()\n\nbest_corrects = 0\n\nfor epoch in range(20):\n    # TRAIN\n    model.train()\n    train_loss, train_corrects = [], []\n    for data, target in train_loader:\n        if torch.cuda.is_available():\n            data, target = data.cuda(), target.cuda()\n\n        optimizer.zero_grad()\n        output = model(data)\n        loss = loss_fn(output, target)\n        loss.backward()\n        optimizer.step()\n\n        preds = output.argmax(dim=1)\n        targets = target.argmax(dim=1)\n        train_corrects.append((preds == targets).sum().item())\n        train_loss.append(loss.item())\n\n    # VALID\n    model.eval()\n    val_loss, val_corrects = [], []\n    with torch.no_grad():\n        for data, target in val_loader:\n            if torch.cuda.is_available():\n                data, target = data.cuda(), target.cuda()\n            output = model(data)\n            loss = loss_fn(output, target)\n            val_loss.append(loss.item())\n\n            preds = output.argmax(dim=1)\n            targets = target.argmax(dim=1)\n            val_corrects.append((preds == targets).sum().item())\n\n    print(f\"Epoch {epoch}: Train Loss={np.mean(train_loss):.4f}, Train Acc={sum(train_corrects)/len(train_dataset):.4f}, \"\n          f\"Val Loss={np.mean(val_loss):.4f}, Val Acc={sum(val_corrects)/len(val_dataset):.4f}\")\n\n    if sum(val_corrects) > best_corrects:\n        best_corrects = sum(val_corrects)\n        WORKING_DIR = '/kaggle/working/'\n        os.makedirs(WORKING_DIR, exist_ok=True)  # на всякий случай\n\n        torch.save(model, os.path.join(WORKING_DIR, 'best_model.pt'))\n\n    scheduler.step()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:20:34.325676Z","iopub.execute_input":"2025-12-05T02:20:34.326206Z","iopub.status.idle":"2025-12-05T02:21:51.516413Z","shell.execute_reply.started":"2025-12-05T02:20:34.326182Z","shell.execute_reply":"2025-12-05T02:21:51.515501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Инференс на тесте и формирование submission.csv","metadata":{}},{"cell_type":"code","source":"import torch\nimport timm\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom skimage.transform import resize\nimport librosa\nimport os\n\n# Функция для загрузки тестового файла и превращения в спектрограммы\ndef load_test_file(f):\n    wav, sr = librosa.load('/kaggle/input/rfcx-species-audio-detection/test/' + f, sr=None)\n    segments = int(np.ceil(len(wav) / length))\n    mel_array = []\n\n    for i in range(segments):\n        if (i + 1) * length > len(wav):\n            slice = wav[len(wav) - length:len(wav)]\n        else:\n            slice = wav[i * length:(i + 1) * length]\n\n        mel_spec = librosa.feature.melspectrogram(\n            y=slice, n_fft=fft, hop_length=hop, sr=sr, fmin=fmin, fmax=fmax, power=1.5\n        )\n        mel_spec = resize(mel_spec, (224, 400))\n        mel_spec = mel_spec - np.min(mel_spec)\n        mel_spec = mel_spec / np.max(mel_spec)\n        mel_spec = np.stack((mel_spec, mel_spec, mel_spec))\n        mel_array.append(mel_spec)\n\n    return mel_array\n\n# Загружаем модель\nmodel = timm.create_model('resnest101e', pretrained=True)\nmodel.fc = torch.nn.Sequential(\n    torch.nn.Linear(2048, 1024),\n    torch.nn.ReLU(),\n    torch.nn.Dropout(p=0.2),\n    torch.nn.Linear(1024, 1024),\n    torch.nn.ReLU(),\n    torch.nn.Dropout(p=0.2),\n    torch.nn.Linear(1024, num_birds)\n)\n\nmodel = torch.load(WORKING_DIR + 'best_model.pt', weights_only=False)\nmodel.eval()\n\nif torch.cuda.is_available():\n    model.cuda()\n\nimport shutil\n\nfor f in os.listdir(WORKING_DIR):\n    if f == 'best_model.pt':   # пропускаем модель\n        continue\n    path = os.path.join(WORKING_DIR, f)\n    if os.path.isfile(path):\n        os.remove(path)\n    elif os.path.isdir(path):\n        shutil.rmtree(path)\n\n# Генерация предсказаний\nresults = []\ntest_files = os.listdir('/kaggle/input/rfcx-species-audio-detection/test/')\n\nfor file_name in tqdm(test_files, desc='Processing test files'):\n    data = load_test_file(file_name)\n    data = torch.tensor(data).float()\n    if torch.cuda.is_available():\n        data = data.cuda()\n\n    output = model(data)\n    maxed_output = torch.max(output, dim=0)[0].cpu().detach().numpy()\n\n    file_id = file_name.split('.')[0]\n    row = [file_id] + maxed_output.tolist()\n    results.append(row)\n\n# Сохраняем в CSV\ncolumns = ['recording_id'] + [f's{i}' for i in range(num_birds)]\nsubmission_df = pd.DataFrame(results, columns=columns)\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"submission.csv создан!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T02:02:36.086971Z","iopub.execute_input":"2025-12-05T02:02:36.087549Z","iopub.status.idle":"2025-12-05T02:05:35.360107Z","shell.execute_reply.started":"2025-12-05T02:02:36.087526Z","shell.execute_reply":"2025-12-05T02:05:35.359114Z"}},"outputs":[],"execution_count":null}]}