{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13436048,"sourceType":"datasetVersion","datasetId":8528233},{"sourceId":613514,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":460961,"modelId":476733}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Подготовка","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nimport numpy as np\nimport librosa\nimport librosa.display\nimport seaborn as sns\nimport os\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\n# Игнорируем предупреждения\n# from warnings import filterwarnings\n# filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:24:38.893342Z","iopub.execute_input":"2025-10-19T17:24:38.893682Z","iopub.status.idle":"2025-10-19T17:24:44.740277Z","shell.execute_reply.started":"2025-10-19T17:24:38.893659Z","shell.execute_reply":"2025-10-19T17:24:44.739515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(librosa.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:23:09.694880Z","iopub.execute_input":"2025-10-19T17:23:09.695245Z","iopub.status.idle":"2025-10-19T17:23:09.701140Z","shell.execute_reply.started":"2025-10-19T17:23:09.695226Z","shell.execute_reply":"2025-10-19T17:23:09.700115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_file_path = '/kaggle/input/freesound-audio-tagging/train.csv'\ntrain_audio_dir_path = '/kaggle/input/freesound-audio-tagging/audio_train'\ntest_csv_file_path = '/kaggle/input/freesound-audio-tagging/test_post_competition.csv'\ntest_audio_dir_path = '/kaggle/input/freesound-audio-tagging/audio_test'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:24:44.741518Z","iopub.execute_input":"2025-10-19T17:24:44.741871Z","iopub.status.idle":"2025-10-19T17:24:44.746778Z","shell.execute_reply.started":"2025-10-19T17:24:44.741852Z","shell.execute_reply":"2025-10-19T17:24:44.745920Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Анализ аудиофайлов\nНайдем распределение величины n_frames - \"ширины\" спектрограммы","metadata":{"execution":{"iopub.status.busy":"2025-10-19T10:44:39.170006Z","iopub.status.idle":"2025-10-19T10:44:39.170363Z","shell.execute_reply.started":"2025-10-19T10:44:39.170207Z","shell.execute_reply":"2025-10-19T10:44:39.170225Z"}}},{"cell_type":"code","source":"def get_spectrogram(y, sr):\n    X = librosa.feature.melspectrogram(y=y, sr=sr)\n    return librosa.amplitude_to_db(np.abs(X))\n\ndef plot_spectrogram(spec_array, title=''):\n    plt.figure(figsize=(14, 3))\n    librosa.display.specshow(spec_array, x_axis='time', y_axis='linear')\n    plt.colorbar(format=\"%+2.f dB\")\n    plt.title(title + ' Спектрограмма. Линейный масштаб')\n    plt.ylabel(\"Частота (Гц)\")\n    plt.xlabel(\"Время (сек.)\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:24:46.591401Z","iopub.execute_input":"2025-10-19T17:24:46.591759Z","iopub.status.idle":"2025-10-19T17:24:46.597664Z","shell.execute_reply.started":"2025-10-19T17:24:46.591736Z","shell.execute_reply":"2025-10-19T17:24:46.596894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# n_frames_list = []\n# for filename in tqdm(os.listdir(train_audio_dir_path), desc='Подсчет n_frames'):\n#     file_path = os.path.join(train_audio_dir_path, filename)\n#     if not os.path.isfile(file_path):\n#         continue\n#\n#     data, sr = librosa.load(file_path)\n#     sp = get_spectrogram(data, sr)\n#     n_frames_list.append(sp.shape[1])\n#\n# plt.hist(n_frames_list, bins=300, color='blue', alpha=0.7)\n# plt.title('Гистограмма для n_frames')\n# plt.xlabel('Число фреймов')\n# plt.ylabel('Количество файлов')\n# plt.grid(True)\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:23:09.759368Z","iopub.execute_input":"2025-10-19T17:23:09.759646Z","iopub.status.idle":"2025-10-19T17:23:09.788194Z","shell.execute_reply.started":"2025-10-19T17:23:09.759622Z","shell.execute_reply":"2025-10-19T17:23:09.787218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import statistics\n#\n# statistics.median([i[1]for i in n_frames_list])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:23:09.788899Z","iopub.execute_input":"2025-10-19T17:23:09.789127Z","iopub.status.idle":"2025-10-19T17:23:09.813194Z","shell.execute_reply.started":"2025-10-19T17:23:09.789109Z","shell.execute_reply":"2025-10-19T17:23:09.811924Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Медианное значение n_frames = 175. Будем приводить к нему все спектрограммы путем урезания / расширения нулями","metadata":{}},{"cell_type":"markdown","source":"## Формирование набора спектрограмм","metadata":{}},{"cell_type":"code","source":"spec_height, spec_width = 128, 175","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:25:26.511530Z","iopub.execute_input":"2025-10-19T17:25:26.511844Z","iopub.status.idle":"2025-10-19T17:25:26.516070Z","shell.execute_reply.started":"2025-10-19T17:25:26.511822Z","shell.execute_reply":"2025-10-19T17:25:26.515076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calc_spectrograms(dir_path: str, save: bool = False, save_filename: str = 'spectrograms.npy'):\n    spectrograms = np.zeros((len(os.listdir(dir_path)), spec_height, spec_width))\n\n    for i, filename in tqdm(enumerate(os.listdir(dir_path)), desc='Построение спектрограмм'):\n        file_path = os.path.join(dir_path, filename)\n        if not os.path.isfile(file_path):\n            continue\n    \n        data, sr = librosa.load(file_path)\n        spec = get_spectrogram(data, sr)\n        min_height = min(spectrograms.shape[1], spec.shape[0])\n        min_width = min(spectrograms.shape[2], spec.shape[1])\n        spectrograms[i][:min_height, :min_width] = spec[:min_height, :min_width]\n\n    if save:\n        np.save(save_filename, spectrograms)\n\n    return spectrograms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:24:54.947973Z","iopub.execute_input":"2025-10-19T17:24:54.948258Z","iopub.status.idle":"2025-10-19T17:24:54.955024Z","shell.execute_reply.started":"2025-10-19T17:24:54.948239Z","shell.execute_reply":"2025-10-19T17:24:54.954065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"spectrograms = None\nspec_dataset_filename = '/kaggle/input/spectrograms/spectrograms.npy'\nspec_filename = 'spectrograms.npy'\n\nif os.path.isfile(spec_dataset_filename):\n    spectrograms = np.load(spec_dataset_filename)\nelif os.path.isfile(spec_filename):\n    spectrograms = np.load(spec_filename)\nelse:\n    spectrograms = calc_spectrograms(\n        dir_path=train_audio_dir_path,\n        save=True,\n        save_filename=spec_filename\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:23:09.858655Z","iopub.execute_input":"2025-10-19T17:23:09.858939Z","execution_failed":"2025-10-19T17:23:31.962Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Загрузка табличных данных","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_file_path)\nlabel_dict = {k: v for k, v in zip(train_df['fname'], train_df['label'])}\nlabel_set = sorted(set(train_df['label']))\nlabel_to_idx = {label: idx for idx, label in enumerate(label_set)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:56:52.662072Z","iopub.execute_input":"2025-10-19T17:56:52.662406Z","iopub.status.idle":"2025-10-19T17:56:52.687266Z","shell.execute_reply.started":"2025-10-19T17:56:52.662382Z","shell.execute_reply":"2025-10-19T17:56:52.686474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\n\nclass SpectrogramDataset(Dataset):\n    def __init__(self, specs, labels):\n        self.specs = specs.astype('float32')\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.specs)\n\n    def __getitem__(self, idx):\n        x = self.specs[idx]\n        x = torch.from_numpy(x).unsqueeze(0)\n        y = self.labels[idx]\n        return x, y","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-19T17:23:31.968Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Построение модели и обучение","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\nclass SimpleCNN(nn.Module):\n    def __init__(self, num_classes):\n        super(SimpleCNN, self).__init__()\n        self.conv1 = nn.Conv2d(1, 16, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(16, 32, kernel_size=3, padding=1)\n\n        self.fc1 = nn.Linear(32 * 32 * 43, 128)\n        self.fc2 = nn.Linear(128, num_classes)\n        self.relu = nn.ReLU()\n\n    def forward(self, x):\n        x = self.relu(self.conv1(x))  # (batch, 16, 128, 175)\n        x = self.pool(x)              # (batch, 16, 64, 87)\n        x = self.relu(self.conv2(x))  # (batch, 32, 64, 87)\n        x = self.pool(x)              # (batch, 32, 32, 43)\n        x = torch.flatten(x, 1)\n        x = self.relu(self.fc1(x))\n        x = self.fc2(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:56:02.523690Z","iopub.execute_input":"2025-10-19T17:56:02.524006Z","iopub.status.idle":"2025-10-19T17:56:02.532415Z","shell.execute_reply.started":"2025-10-19T17:56:02.523984Z","shell.execute_reply":"2025-10-19T17:56:02.531289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = [\n    label_to_idx[label_dict[filename]]\n    for filename in os.listdir(train_audio_dir_path)\n]\nlabels = torch.tensor(labels)\n\ndataset = SpectrogramDataset(spectrograms, labels)\ndataloader = DataLoader(dataset, batch_size=32, shuffle=True)\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = SimpleCNN(num_classes=len(label_set)).to(device)\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    for inputs, targets in dataloader:\n        inputs, targets = inputs.to(device), targets.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n\n    print(f\"Epoch {epoch+1}/{num_epochs}, Loss: {running_loss/len(dataloader):.4f}\")\n\nprint(\"Обучение завершено.\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-19T17:23:31.971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = 'simple_cnn_model.pth'\ntorch.save(model.state_dict(), model_path)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-19T17:23:31.971Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Формирование выходного файла","metadata":{}},{"cell_type":"code","source":"test_spectrograms = None\ntest_spec_dataset_filename = '/kaggle/input/spectrograms/test_spectrograms.npy'\ntest_spec_filename = 'test_spectrograms.npy'\n\nif os.path.isfile(test_spec_dataset_filename):\n    test_spectrograms = np.load(test_spec_dataset_filename)\nelif os.path.isfile(test_spec_filename):\n    test_spectrograms = np.load(test_spec_filename)\nelse:\n    test_spectrograms = calc_spectrograms(\n        dir_path=test_audio_dir_path,\n        save=True,\n        save_filename=test_spec_filename\n    )","metadata":{"execution":{"iopub.status.busy":"2025-10-19T17:35:32.517996Z","iopub.execute_input":"2025-10-19T17:35:32.518926Z","iopub.status.idle":"2025-10-19T17:40:08.630415Z","shell.execute_reply.started":"2025-10-19T17:35:32.518890Z","shell.execute_reply":"2025-10-19T17:40:08.629401Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(test_csv_file_path)\ntest_label_dict = {k: v for k, v in zip(test_df['fname'], test_df['label'])}\ntest_label_set = sorted(set(train_df['label']))\ntest_label_to_idx = {label: idx for idx, label in enumerate(test_label_set)}\nidx_to_label = {idx: label for label, idx in test_label_to_idx.items()}\nlabels_array = [idx_to_label[i] for i in range(len(test_label_to_idx))]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:56:57.119260Z","iopub.execute_input":"2025-10-19T17:56:57.119922Z","iopub.status.idle":"2025-10-19T17:56:57.142938Z","shell.execute_reply.started":"2025-10-19T17:56:57.119894Z","shell.execute_reply":"2025-10-19T17:56:57.141789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = '/kaggle/input/audio-cnn/pytorch/default/1/simple_cnn_model.pth'\nmodel_loaded = SimpleCNN(num_classes=len(label_set))\nmodel_loaded.load_state_dict(torch.load(model_path, map_location=torch.device('cpu')))\nmodel_loaded.to('cpu')\nmodel_loaded.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:57:00.838386Z","iopub.execute_input":"2025-10-19T17:57:00.838762Z","iopub.status.idle":"2025-10-19T17:57:01.358998Z","shell.execute_reply.started":"2025-10-19T17:57:00.838739Z","shell.execute_reply":"2025-10-19T17:57:01.358169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nmodel_loaded.eval()\ncorrect = 0\ntotal = 0\nout_labels = [0] * len(test_spectrograms)\nwith torch.no_grad():\n    for i, spec in enumerate(test_spectrograms):\n        spec_tensor = torch.from_numpy(spec.astype('float32')).unsqueeze(0).unsqueeze(0)\n        out = model_loaded(spec_tensor)\n        _, max_prob_idx = torch.max(out, 1)\n        out_labels[i] = labels_array[max_prob_idx.item()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:57:04.921303Z","iopub.execute_input":"2025-10-19T17:57:04.922036Z","iopub.status.idle":"2025-10-19T17:57:43.344474Z","shell.execute_reply.started":"2025-10-19T17:57:04.922005Z","shell.execute_reply":"2025-10-19T17:57:43.343694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"out_filenames = [filename for filename in os.listdir(test_audio_dir_path)]\nprint(len(out_filenames), len(out_labels))\nout_df = pd.DataFrame({\n    'fname': out_filenames,\n    'label': out_labels\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:57:43.345854Z","iopub.execute_input":"2025-10-19T17:57:43.346188Z","iopub.status.idle":"2025-10-19T17:57:43.357462Z","shell.execute_reply.started":"2025-10-19T17:57:43.346159Z","shell.execute_reply":"2025-10-19T17:57:43.356495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"out_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-19T17:58:06.995264Z","iopub.execute_input":"2025-10-19T17:58:06.996152Z","iopub.status.idle":"2025-10-19T17:58:07.025055Z","shell.execute_reply.started":"2025-10-19T17:58:06.996115Z","shell.execute_reply":"2025-10-19T17:58:07.023397Z"}},"outputs":[],"execution_count":null}]}