{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\nimport numpy as np\n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data import DataLoader\nfrom torchvision import transforms\nfrom torchvision.models import efficientnet_b0, EfficientNet_B0_Weights\n\n#from warnings import filterwarnings\n#filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-25T14:40:45.995302Z","iopub.execute_input":"2023-12-25T14:40:45.995780Z","iopub.status.idle":"2023-12-25T14:40:48.934477Z","shell.execute_reply.started":"2023-12-25T14:40:45.995738Z","shell.execute_reply":"2023-12-25T14:40:48.933094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:48.936769Z","iopub.execute_input":"2023-12-25T14:40:48.937345Z","iopub.status.idle":"2023-12-25T14:40:48.943350Z","shell.execute_reply.started":"2023-12-25T14:40:48.937289Z","shell.execute_reply":"2023-12-25T14:40:48.942136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/freesound-audio-tagging/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:48.944617Z","iopub.execute_input":"2023-12-25T14:40:48.945473Z","iopub.status.idle":"2023-12-25T14:40:48.982608Z","shell.execute_reply.started":"2023-12-25T14:40:48.945430Z","shell.execute_reply":"2023-12-25T14:40:48.981218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Как и сказано в описании данных, есть всего 41 категория. Отсортируем их и закодируем.","metadata":{}},{"cell_type":"code","source":"labels = np.unique(train.label.values)\nlabel_encoder = {label:i for i, label in enumerate(labels)}\nprint(label_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:48.985597Z","iopub.execute_input":"2023-12-25T14:40:48.986110Z","iopub.status.idle":"2023-12-25T14:40:48.998681Z","shell.execute_reply.started":"2023-12-25T14:40:48.986060Z","shell.execute_reply":"2023-12-25T14:40:48.997389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Разделим тренировочную выборку на \"тренировочную 2\" и валидационную.","metadata":{}},{"cell_type":"code","source":"train, validation = train_test_split(train, test_size=0.2, shuffle=True, random_state=5)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:49.000015Z","iopub.execute_input":"2023-12-25T14:40:49.000496Z","iopub.status.idle":"2023-12-25T14:40:49.012373Z","shell.execute_reply.started":"2023-12-25T14:40:49.000447Z","shell.execute_reply":"2023-12-25T14:40:49.010749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Для получения акустических признаков используем спектограмму Мела.","metadata":{}},{"cell_type":"code","source":"TRAIN_PATH = '../input/freesound-audio-tagging/audio_train/'\nTEST_PATH = '../input/freesound-audio-tagging/audio_test/'\n\nclass Dataset(Dataset):\n    def __init__(self, dataframe, test=False):\n        self.dataframe = dataframe\n        self.test = test\n        \n    def __getitem__(self, index):\n        fileName = self.dataframe.fname.values[index]\n        label = self.dataframe.label.values[index]\n        \n        path = (TEST_PATH if self.test else TRAIN_PATH) + fileName\n        signal, _ = librosa.load(path)\n        signal = librosa.feature.melspectrogram(y=signal)    \n        signal = librosa.power_to_db(signal, ref=np.max) \n        \n        try:\n            resized = cv2.resize(signal, (128, 128))\n        except Exception as e:\n            print(path)\n            print(str(e))\n            resized = np.zeros(shape=(128, 128))\n        \n        x = np.stack([resized] * 3)\n        x = torch.tensor(x, dtype=torch.float32)\n\n        if self.test == False:\n            y = label_encoder[label]\n            return x, y\n        else:\n             return x\n        \n    def __len__(self):\n        return self.dataframe.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:49.014731Z","iopub.execute_input":"2023-12-25T14:40:49.015290Z","iopub.status.idle":"2023-12-25T14:40:49.026015Z","shell.execute_reply.started":"2023-12-25T14:40:49.015252Z","shell.execute_reply":"2023-12-25T14:40:49.024796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Разбиваем выборки на пакеты по 64 для предстоящих \"mini-batch\" проходов.","metadata":{}},{"cell_type":"code","source":"batch_size = 64\n\ntrain_set = Dataset(train)\nval_set = Dataset(validation)\ntrain_loader = DataLoader(train_set, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_set , batch_size=batch_size, shuffle=True)\nprint('Train set: {}, Validation set: {}'.format(train.shape[0], validation.shape[0]))","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:49.027685Z","iopub.execute_input":"2023-12-25T14:40:49.028428Z","iopub.status.idle":"2023-12-25T14:40:49.037776Z","shell.execute_reply.started":"2023-12-25T14:40:49.028385Z","shell.execute_reply":"2023-12-25T14:40:49.036403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Загрузим предобученную сверточную сеть efficientnet_b0. Изменяем классификатор под нашу задачу (41 категория) и выбираем тип устройства для вычислений.","metadata":{}},{"cell_type":"code","source":"model = efficientnet_b0(weights='EfficientNet_B0_Weights.DEFAULT')\nmodel.classifier[1] = torch.nn.Linear(1280, 41)\nmodel = model.to(device)\nmodel.to(device);","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:49.039319Z","iopub.execute_input":"2023-12-25T14:40:49.039715Z","iopub.status.idle":"2023-12-25T14:40:49.260543Z","shell.execute_reply.started":"2023-12-25T14:40:49.039678Z","shell.execute_reply":"2023-12-25T14:40:49.259515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Для обучения будем использовать оптимизатор AdamW, а в качестве функции потерь - категориальную кросс-энтропию, которая является наиболее подходящей для модели классификации.","metadata":{}},{"cell_type":"code","source":"epochs = 10\noptimizer = torch.optim.AdamW(model.parameters(), lr=0.001)\ncost = torch.nn.CrossEntropyLoss()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:49.261974Z","iopub.execute_input":"2023-12-25T14:40:49.262389Z","iopub.status.idle":"2023-12-25T14:40:49.269678Z","shell.execute_reply.started":"2023-12-25T14:40:49.262352Z","shell.execute_reply":"2023-12-25T14:40:49.268724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in range(epochs):\n    train_loss = 0\n    val_loss = 0\n    train_correct = 0\n    val_correct = 0\n    \n    # перевод модели в режим обучения\n    model.train()\n    for x, y in train_loader:\n        optimizer.zero_grad()\n        x,y = x.to(device),y.to(device)\n        \n        # вычисление предсказания и потерь\n        pred = model(x)\n        loss = cost(pred, y)\n        train_loss += cost(pred, y).item()\n        train_correct += (pred.argmax(1) == y).type(torch.float).sum().item()\n        \n        # обратное распространение ошибки\n        loss.backward()\n        optimizer.step()\n    \n    # перевод модели в режим оценивания\n    model.eval()\n    with torch.no_grad():\n        for x, y in val_loader:\n            x,y = x.to(device),y.to(device)\n            \n            pred = model(x)\n            loss = cost(pred, y)\n            val_loss += cost(pred, y).item()\n            val_correct += (pred.argmax(1) == y).type(torch.float).sum().item()\n            \n    train_loss = train_loss/len(train_loader)\n    val_loss = val_loss/len(val_loader)\n    train_accuracy = train_correct / len(train)\n    val_accuracy = val_correct / len(validation)\n    print(\"epoch = %d, train_loss = %.5f, val_loss = %.5f, train_accuracy = %.5f, val_accuracy = %.5f\" % (epoch, train_loss, val_loss, train_accuracy, val_accuracy))","metadata":{"execution":{"iopub.status.busy":"2023-12-25T14:40:49.272601Z","iopub.execute_input":"2023-12-25T14:40:49.273308Z","iopub.status.idle":"2023-12-25T16:37:31.491463Z","shell.execute_reply.started":"2023-12-25T14:40:49.273266Z","shell.execute_reply":"2023-12-25T16:37:31.488414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Вычисляем предсказания для тестовой выборки.","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('../input/freesound-audio-tagging/sample_submission.csv')\n\ntest_dataset = Dataset(test, test=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\npredictions = torch.tensor([])\nmodel.eval()\nfor x in test_loader:\n    x = x.to(device)\n    with torch.no_grad():\n        y_hat = model(x)\n    predictions = torch.cat([predictions, y_hat.cpu()])","metadata":{"execution":{"iopub.status.busy":"2023-12-25T17:07:09.449523Z","iopub.execute_input":"2023-12-25T17:07:09.451526Z","iopub.status.idle":"2023-12-25T17:16:56.853804Z","shell.execute_reply.started":"2023-12-25T17:07:09.451438Z","shell.execute_reply":"2023-12-25T17:16:56.852344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Используем функцию активации softmax для получения вектора вероятностей (вероятностей принадлежности к каждой категории звуков).","metadata":{}},{"cell_type":"code","source":"predictions = torch.nn.functional.softmax(predictions, dim=1).detach().numpy()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T17:22:47.739440Z","iopub.execute_input":"2023-12-25T17:22:47.739996Z","iopub.status.idle":"2023-12-25T17:22:47.750828Z","shell.execute_reply.started":"2023-12-25T17:22:47.739948Z","shell.execute_reply":"2023-12-25T17:22:47.749602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Расставляем метки для тестового набора.","metadata":{}},{"cell_type":"code","source":"submission_top1 = test.copy()\n\nN = len(test)\nfor i in range(N):\n    p = predictions[i, :]\n    idx = np.argmax(p)\n    submission_top1.label[i] = labels[idx]\n\nsubmission_top1.to_csv('submission_final.csv', index=False, header=True)\n\nsubmission_top1.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T17:24:33.156222Z","iopub.execute_input":"2023-12-25T17:24:33.157287Z","iopub.status.idle":"2023-12-25T17:24:34.384011Z","shell.execute_reply.started":"2023-12-25T17:24:33.157234Z","shell.execute_reply":"2023-12-25T17:24:34.382814Z"},"trusted":true},"execution_count":null,"outputs":[]}]}