{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport cv2\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom torchvision import models\nfrom IPython.display import Audio, display","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T19:26:42.833021Z","iopub.execute_input":"2025-10-18T19:26:42.833831Z","iopub.status.idle":"2025-10-18T19:26:42.838292Z","shell.execute_reply.started":"2025-10-18T19:26:42.833804Z","shell.execute_reply":"2025-10-18T19:26:42.837635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T19:26:46.218948Z","iopub.execute_input":"2025-10-18T19:26:46.219723Z","iopub.status.idle":"2025-10-18T19:26:46.223867Z","shell.execute_reply.started":"2025-10-18T19:26:46.219694Z","shell.execute_reply":"2025-10-18T19:26:46.223118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Загрузка датасета\ntrain_path = '../input/freesound-audio-tagging/audio_train/'\ntrain_df = pd.read_csv(\"../input/freesound-audio-tagging/train.csv\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T19:26:48.135350Z","iopub.execute_input":"2025-10-18T19:26:48.136149Z","iopub.status.idle":"2025-10-18T19:26:48.152902Z","shell.execute_reply.started":"2025-10-18T19:26:48.136121Z","shell.execute_reply":"2025-10-18T19:26:48.152278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Выводим пример звукового файла\nfname = train_df.iloc[25]['fname']\nlabel = train_df.iloc[25]['label']\nprint(f\"Метка: {label}\")\n\naudio_path = os.path.join(train_path, fname)\nAudio(filename=audio_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:12:30.215447Z","iopub.execute_input":"2025-10-18T18:12:30.216042Z","iopub.status.idle":"2025-10-18T18:12:30.242825Z","shell.execute_reply.started":"2025-10-18T18:12:30.216014Z","shell.execute_reply":"2025-10-18T18:12:30.242061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioDataset(Dataset):\n    def __init__(self, df, test=False):\n        self.df = df\n        self.test = test\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        fname = self.df.iloc[idx]['fname']\n        path = (test_path if self.test else train_path) + fname\n        \n        signal, _ = librosa.load(path, sr=22050, duration=4.0)\n        if len(signal) < 22050 * 4:\n            signal = np.pad(signal, (0, 22050 * 4 - len(signal)), mode='constant')\n        \n        mel = librosa.feature.melspectrogram(y=signal, sr=22050, n_mels=128, fmax=11025)\n        mel_db = librosa.power_to_db(mel, ref=np.max)\n        \n        try:\n            mel_resized = cv2.resize(mel_db, IMG_SIZE[::-1])\n        except:\n            mel_resized = np.zeros(IMG_SIZE)\n        \n        X = np.stack([mel_resized] * 3, axis=0)\n\n        if self.test:\n            return torch.tensor(X, dtype=torch.float32)\n        else:\n            label = self.df.iloc[idx]['label']\n            y = label_to_idx[label]\n            return torch.tensor(X, dtype=torch.float32), y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:12:41.201477Z","iopub.execute_input":"2025-10-18T18:12:41.202019Z","iopub.status.idle":"2025-10-18T18:12:41.209269Z","shell.execute_reply.started":"2025-10-18T18:12:41.201978Z","shell.execute_reply":"2025-10-18T18:12:41.208455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Всего обучающих примеров: {len(train_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:47:34.287225Z","iopub.execute_input":"2025-10-18T18:47:34.287533Z","iopub.status.idle":"2025-10-18T18:47:34.291732Z","shell.execute_reply.started":"2025-10-18T18:47:34.287512Z","shell.execute_reply":"2025-10-18T18:47:34.290955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Оставим только вручную проверенные (более надёжные)\nclean_df = train_df[train_df['manually_verified'] == 1].reset_index(drop=True)\nprint(f\"Проверенные примеры: {len(clean_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:12:49.173913Z","iopub.execute_input":"2025-10-18T18:12:49.174503Z","iopub.status.idle":"2025-10-18T18:12:49.183724Z","shell.execute_reply.started":"2025-10-18T18:12:49.174481Z","shell.execute_reply":"2025-10-18T18:12:49.182904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Разделение\ntrain_data, val_data = train_test_split(\n    clean_df, test_size=0.2, stratify=clean_df['label'], random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:13:18.776582Z","iopub.execute_input":"2025-10-18T18:13:18.777149Z","iopub.status.idle":"2025-10-18T18:13:18.786851Z","shell.execute_reply.started":"2025-10-18T18:13:18.777125Z","shell.execute_reply":"2025-10-18T18:13:18.786078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Размер тренировочного датасета: {len(train_data)}\")\nprint(f\"Размер валидационного датасета: {len(val_data)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:50:10.232857Z","iopub.execute_input":"2025-10-18T18:50:10.233441Z","iopub.status.idle":"2025-10-18T18:50:10.237410Z","shell.execute_reply.started":"2025-10-18T18:50:10.233417Z","shell.execute_reply":"2025-10-18T18:50:10.236691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Кодирование меток\nunique_labels = sorted(clean_df['label'].unique())\nlabel_to_idx = {label: idx for idx, label in enumerate(unique_labels)}\nidx_to_label = {idx: label for label, idx in label_to_idx.items()}\nnum_classes = len(unique_labels)\nprint(f\"Число классов: {num_classes}\")\nidx_to_label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:50:18.540468Z","iopub.execute_input":"2025-10-18T18:50:18.541034Z","iopub.status.idle":"2025-10-18T18:50:18.548352Z","shell.execute_reply.started":"2025-10-18T18:50:18.541003Z","shell.execute_reply":"2025-10-18T18:50:18.547586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = (128, 128)\nbatch_size = 32","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:13:30.098327Z","iopub.execute_input":"2025-10-18T18:13:30.098990Z","iopub.status.idle":"2025-10-18T18:13:30.103003Z","shell.execute_reply.started":"2025-10-18T18:13:30.098955Z","shell.execute_reply":"2025-10-18T18:13:30.102059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = DataLoader(AudioDataset(train_data), batch_size=batch_size, shuffle=True, num_workers=0)\nval_loader = DataLoader(AudioDataset(val_data), batch_size=batch_size, shuffle=False, num_workers=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:13:40.390103Z","iopub.execute_input":"2025-10-18T18:13:40.390610Z","iopub.status.idle":"2025-10-18T18:13:40.395141Z","shell.execute_reply.started":"2025-10-18T18:13:40.390588Z","shell.execute_reply":"2025-10-18T18:13:40.394354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Модель\nfrom torchvision.models import EfficientNet_B0_Weights\nweights = EfficientNet_B0_Weights.DEFAULT\nmodel = models.efficientnet_b0(weights=weights)\nmodel.classifier[1] = nn.Linear(model.classifier[1].in_features, num_classes)\nmodel = model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T19:26:55.551461Z","iopub.execute_input":"2025-10-18T19:26:55.552016Z","iopub.status.idle":"2025-10-18T19:26:55.693680Z","shell.execute_reply.started":"2025-10-18T19:26:55.551985Z","shell.execute_reply":"2025-10-18T19:26:55.693081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обучение\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='max', patience=2, factor=0.5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:13:47.007377Z","iopub.execute_input":"2025-10-18T18:13:47.007958Z","iopub.status.idle":"2025-10-18T18:13:47.013185Z","shell.execute_reply.started":"2025-10-18T18:13:47.007934Z","shell.execute_reply":"2025-10-18T18:13:47.012335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# МЕТРИКА MAP@3\ndef map_at_3(y_true, y_pred_probs):\n    \"\"\"\n    y_true: массив истинных меток (целые числа)\n    y_pred_probs: матрица вероятностей (N x num_classes)\n    Возвращает MAP@3\n    \"\"\"\n    top3_preds = np.argsort(y_pred_probs, axis=1)[:, ::-1][:, :3]  # (N, 3)\n    scores = []\n    for i, true_label in enumerate(y_true):\n        if true_label in top3_preds[i]:\n            rank = np.where(top3_preds[i] == true_label)[0][0] + 1  # 1-based\n            scores.append(1.0 / rank)\n        else:\n            scores.append(0.0)\n    return np.mean(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:13:49.308295Z","iopub.execute_input":"2025-10-18T18:13:49.308582Z","iopub.status.idle":"2025-10-18T18:13:49.313787Z","shell.execute_reply.started":"2025-10-18T18:13:49.308562Z","shell.execute_reply":"2025-10-18T18:13:49.312902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_map3 = 0.0\nfor epoch in range(15):\n    model.train()\n    for x, y in train_loader:\n        x, y = x.to(device), y.to(device)\n        optimizer.zero_grad()\n        out = model(x)\n        loss = criterion(out, y)\n        loss.backward()\n        optimizer.step()\n    \n    # Валидация по MAP@3\n    model.eval()\n    all_probs = []\n    all_true = []\n    with torch.no_grad():\n        for x, y in val_loader:\n            x, y = x.to(device), y.to(device)\n            logits = model(x)\n            probs = F.softmax(logits, dim=1).cpu().numpy()\n            all_probs.append(probs)\n            all_true.append(y.cpu().numpy())\n    \n    all_probs = np.vstack(all_probs)\n    all_true = np.concatenate(all_true)\n    val_map3 = map_at_3(all_true, all_probs)\n    print(f\"Epoch {epoch+1}: Val MAP@3 = {val_map3:.4f}\")\n    \n    if val_map3 > best_map3:\n        best_map3 = val_map3\n        torch.save(model.state_dict(), \"/kaggle/working/best_model.pth\")\n    scheduler.step(val_map3)\n\nprint(f\"Лучший валидационный MAP@3: {best_map3:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:13:57.553004Z","iopub.execute_input":"2025-10-18T18:13:57.553630Z","iopub.status.idle":"2025-10-18T18:26:29.065878Z","shell.execute_reply.started":"2025-10-18T18:13:57.553606Z","shell.execute_reply":"2025-10-18T18:26:29.065065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_path = '../input/freesound-audio-tagging/audio_test/'\ntest_df = pd.read_csv('../input/freesound-audio-tagging/sample_submission.csv')\ntest_loader = DataLoader(AudioDataset(test_df, test=True), batch_size=batch_size, shuffle=False, num_workers=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T19:19:23.214165Z","iopub.execute_input":"2025-10-18T19:19:23.214873Z","iopub.status.idle":"2025-10-18T19:19:23.230075Z","shell.execute_reply.started":"2025-10-18T19:19:23.214849Z","shell.execute_reply":"2025-10-18T19:19:23.229251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Размер  датасета: {len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:59:26.822093Z","iopub.execute_input":"2025-10-18T18:59:26.822671Z","iopub.status.idle":"2025-10-18T18:59:26.826417Z","shell.execute_reply.started":"2025-10-18T18:59:26.822649Z","shell.execute_reply":"2025-10-18T18:59:26.825753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Предсказание на тесте (топ 3 метки)\nmodel.load_state_dict(torch.load(\"/kaggle/working/best_model.pth\"))\nmodel.eval()\n\nall_probs = []\nwith torch.no_grad():\n    for x in test_loader:\n        x = x.to(device)\n        logits = model(x)\n        probs = F.softmax(logits, dim=1).cpu().numpy()\n        all_probs.append(probs)\n\nall_probs = np.vstack(all_probs)\n\n# Формируем решение\nsubmission = []\nfor i, fname in enumerate(test_df['fname']):\n    top3_idx = np.argsort(all_probs[i])[::-1][:3]\n    top3_labels = [idx_to_label[idx] for idx in top3_idx]\n    submission.append({'fname': fname, 'label': ' '.join(top3_labels)})\n\n# Преобразуем в DataFrame\nsubmission_df = pd.DataFrame(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:31:48.832045Z","iopub.execute_input":"2025-10-18T18:31:48.832344Z","iopub.status.idle":"2025-10-18T18:33:27.029024Z","shell.execute_reply.started":"2025-10-18T18:31:48.832320Z","shell.execute_reply":"2025-10-18T18:33:27.028195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Пример 5 предсказаний\nprint(\"Примеры предсказаний на тестовых данных:\\n\")\n\nnum_examples = 5\nfor i in range(num_examples):\n    fname = submission_df.iloc[i]['fname']\n    pred_labels = submission_df.iloc[i]['label']\n    \n    print(f\"Файл: {fname}\")\n    print(f\"Предсказанные метки (top-3): {pred_labels}\")\n    \n    # Путь к аудиофайлу\n    audio_path = os.path.join(test_path, fname)\n    \n    # Отображаем аудиоплеер\n    display(Audio(filename=audio_path))\n    print(\"-\" * 60)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Сохраняем решение\npd.DataFrame(submission).to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:40:34.881151Z","iopub.execute_input":"2025-10-18T18:40:34.881467Z","iopub.status.idle":"2025-10-18T18:40:34.911683Z","shell.execute_reply.started":"2025-10-18T18:40:34.881442Z","shell.execute_reply":"2025-10-18T18:40:34.910946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Проверка формата\ncheck = pd.read_csv(\"submission.csv\")\nprint(check.head(10))\nprint(\"\\nРазмер:\", check.shape)\nprint(\"Колонки:\", check.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T18:40:50.417463Z","iopub.execute_input":"2025-10-18T18:40:50.417776Z","iopub.status.idle":"2025-10-18T18:40:50.437142Z","shell.execute_reply.started":"2025-10-18T18:40:50.417756Z","shell.execute_reply":"2025-10-18T18:40:50.436460Z"}},"outputs":[],"execution_count":null}]}