{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.426696Z","iopub.execute_input":"2023-11-13T08:56:21.427552Z","iopub.status.idle":"2023-11-13T08:56:21.432024Z","shell.execute_reply.started":"2023-11-13T08:56:21.427521Z","shell.execute_reply":"2023-11-13T08:56:21.431068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Загружаем датасет и выводим его:\ntrain_path = '../input/freesound-audio-tagging/audio_train/'\ntrain = pd.read_csv(\"../input/freesound-audio-tagging/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.433738Z","iopub.execute_input":"2023-11-13T08:56:21.434054Z","iopub.status.idle":"2023-11-13T08:56:21.458258Z","shell.execute_reply.started":"2023-11-13T08:56:21.434030Z","shell.execute_reply":"2023-11-13T08:56:21.457517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# оцениваем, какие есть классы\nunique_labels = train.label.unique()\nprint(\"Labels:\", unique_labels)\nlabels = np.unique(train.label.values)\nlabel_encoder = {label:i for i, label in enumerate(labels)}","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.459707Z","iopub.execute_input":"2023-11-13T08:56:21.459957Z","iopub.status.idle":"2023-11-13T08:56:21.474265Z","shell.execute_reply.started":"2023-11-13T08:56:21.459935Z","shell.execute_reply":"2023-11-13T08:56:21.473360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision\n# скачиваем предобученную модель efficientnet с наилучшими весами:\nmodel = torchvision.models.efficientnet_b0(weights='EfficientNet_B0_Weights.DEFAULT')","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.475439Z","iopub.execute_input":"2023-11-13T08:56:21.475700Z","iopub.status.idle":"2023-11-13T08:56:21.635922Z","shell.execute_reply.started":"2023-11-13T08:56:21.475677Z","shell.execute_reply":"2023-11-13T08:56:21.634953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n# определяем тип девайса, на котором будут проводиться вычисления\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.638153Z","iopub.execute_input":"2023-11-13T08:56:21.638442Z","iopub.status.idle":"2023-11-13T08:56:21.643673Z","shell.execute_reply.started":"2023-11-13T08:56:21.638417Z","shell.execute_reply":"2023-11-13T08:56:21.642844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Производим замену классификатора на нужный под нашу задачу\nmodel.classifier[1] = torch.nn.Linear(1280, 41)\nmodel.to(device)\n# Подбираем оптимизатор и критерий останова под задачу классификации\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)\ncriterion = torch.nn.CrossEntropyLoss()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-13T08:56:21.644808Z","iopub.execute_input":"2023-11-13T08:56:21.645096Z","iopub.status.idle":"2023-11-13T08:56:21.674593Z","shell.execute_reply.started":"2023-11-13T08:56:21.645073Z","shell.execute_reply":"2023-11-13T08:56:21.673746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\nimport librosa\nimport cv2\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.675666Z","iopub.execute_input":"2023-11-13T08:56:21.675920Z","iopub.status.idle":"2023-11-13T08:56:21.680524Z","shell.execute_reply.started":"2023-11-13T08:56:21.675898Z","shell.execute_reply":"2023-11-13T08:56:21.679719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = train_path\nTEST_PATH = '../input/freesound-audio-tagging/audio_test/'\nbatch_size = 64\nepochs = 10\n\nclass SoundDataset(Dataset):\n    def __init__(self, dataframe, path, test=False):\n        super(SoundDataset, self).__init__()\n        self.dataframe = dataframe\n        self.path = path\n        self.test = test\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        file_path = self.dataframe.fname.values[idx]\n        label = self.dataframe.label.values[idx]\n        path = (TEST_PATH if self.test else TRAIN_PATH) + file_path\n        signal, _ = librosa.load(path)\n        signal = librosa.feature.melspectrogram(y=signal)\n        signal = librosa.power_to_db(signal, ref=np.max)\n\n        try:\n            resized = cv2.resize(signal, (128, 128))\n        except Exception as e:\n            print(path)\n            print(str(e))\n            resized = np.zeros(shape=(128, 128))\n\n        X = np.stack([resized] * 3)  # Дублирование каналов\n        X = torch.tensor(X, dtype=torch.float32)\n\n        if not self.test:\n            y = label_encoder[label]\n            return X, y\n        else:\n            return X","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.681630Z","iopub.execute_input":"2023-11-13T08:56:21.681908Z","iopub.status.idle":"2023-11-13T08:56:21.691903Z","shell.execute_reply.started":"2023-11-13T08:56:21.681885Z","shell.execute_reply":"2023-11-13T08:56:21.691091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_validation, y_train, y_validation = train_test_split(train, train, test_size=0.2, shuffle=True, random_state=5)\ntrain_dataset = SoundDataset(x_train, TRAIN_PATH)\nval_dataset = SoundDataset(x_validation, TRAIN_PATH)\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.692950Z","iopub.execute_input":"2023-11-13T08:56:21.693209Z","iopub.status.idle":"2023-11-13T08:56:21.710673Z","shell.execute_reply.started":"2023-11-13T08:56:21.693177Z","shell.execute_reply":"2023-11-13T08:56:21.709859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in range(epochs):\n    model.train()\n    train_loss = 0\n    train_correct = 0\n    for X, y in train_loader:\n        X, y = X.to(device), y.to(device)\n        optimizer.zero_grad()\n        pred = model(X)\n        loss = criterion(pred, y)\n        loss.backward()\n        optimizer.step()\n        train_loss += loss.item() * X.size(0)\n        train_correct += torch.sum(pred.argmax(1) == y).item()\n\n    train_loss = train_loss / len(train_loader.dataset)\n    train_accuracy = train_correct / len(train_loader.dataset)\n\n    model.eval()\n    val_loss = 0\n    val_correct = 0\n    with torch.no_grad():\n        for X, y in val_loader:\n            X, y = X.to(device), y.to(device)\n            pred = model(X)\n            loss = criterion(pred, y)\n            val_loss += loss.item() * X.size(0)\n            val_correct += torch.sum(pred.argmax(1) == y).item()\n\n    val_loss = val_loss / len(val_loader.dataset)\n    val_accuracy = val_correct / len(val_loader.dataset)\n\n    print(\"Epoch {}, Train Loss: {:.4f}, Train Accuracy: {:.4f}, Val Loss: {:.4f}, Val Accuracy: {:.4f}\".format(epoch+1, train_loss, train_accuracy, val_loss, val_accuracy))","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:21.711830Z","iopub.execute_input":"2023-11-13T08:56:21.712181Z","iopub.status.idle":"2023-11-13T08:56:26.166821Z","shell.execute_reply.started":"2023-11-13T08:56:21.712157Z","shell.execute_reply":"2023-11-13T08:56:26.165030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/freesound-audio-tagging/sample_submission.csv')\ntest_dataset = SoundDataset(test, TEST_PATH, test=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\npredictions = torch.tensor([])","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:26.167648Z","iopub.status.idle":"2023-11-13T08:56:26.167976Z","shell.execute_reply.started":"2023-11-13T08:56:26.167816Z","shell.execute_reply":"2023-11-13T08:56:26.167831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()\nwith torch.no_grad():\n    for X in test_loader:\n        X = X.to(device)\n        y_hat = model(X)\n        predictions = torch.cat([predictions, y_hat.cpu()])\n\npredictions = F.softmax(predictions, dim=1).detach().numpy()\n\nsubmission_top1 = test.copy()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:26.169134Z","iopub.status.idle":"2023-11-13T08:56:26.169633Z","shell.execute_reply.started":"2023-11-13T08:56:26.169378Z","shell.execute_reply":"2023-11-13T08:56:26.169403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(test)):\n    p = predictions[i, :]\n    idx = np.argmax(p)\n    submission_top1.label[i] = labels[idx]\n\nsubmission_top1.to_csv('submission_final.csv', index=False, header=True)\n\nsubmission_top1.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:56:26.170897Z","iopub.status.idle":"2023-11-13T08:56:26.171334Z","shell.execute_reply.started":"2023-11-13T08:56:26.171108Z","shell.execute_reply":"2023-11-13T08:56:26.171130Z"},"trusted":true},"execution_count":null,"outputs":[]}]}