{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n# Загружаем датасет и выводим его:\ntrain_path = '../input/freesound-audio-tagging/audio_train/'\ntrain = pd.read_csv(\"../input/freesound-audio-tagging/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.529318Z","iopub.execute_input":"2024-10-21T17:56:52.529848Z","iopub.status.idle":"2024-10-21T17:56:52.554230Z","shell.execute_reply.started":"2024-10-21T17:56:52.529798Z","shell.execute_reply":"2024-10-21T17:56:52.553188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# оцениваем, какие есть классы\nunique_labels = train.label.unique()\nprint(\"Labels:\", unique_labels)\nlabels = np.unique(train.label.values)\nlabel_encoder = {label:i for i, label in enumerate(labels)}","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.556324Z","iopub.execute_input":"2024-10-21T17:56:52.556659Z","iopub.status.idle":"2024-10-21T17:56:52.574620Z","shell.execute_reply.started":"2024-10-21T17:56:52.556624Z","shell.execute_reply":"2024-10-21T17:56:52.573223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision\n# скачиваем предобученную модель efficientnet с наилучшими весами:\nmodel = torchvision.models.efficientnet_b0(weights='EfficientNet_B0_Weights.DEFAULT')","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.576255Z","iopub.execute_input":"2024-10-21T17:56:52.576652Z","iopub.status.idle":"2024-10-21T17:56:52.769535Z","shell.execute_reply.started":"2024-10-21T17:56:52.576613Z","shell.execute_reply":"2024-10-21T17:56:52.768254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n# определяем тип девайса, на котором будут проводиться вычисления\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.771457Z","iopub.execute_input":"2024-10-21T17:56:52.771871Z","iopub.status.idle":"2024-10-21T17:56:52.779253Z","shell.execute_reply.started":"2024-10-21T17:56:52.771829Z","shell.execute_reply":"2024-10-21T17:56:52.777658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Производим замену классификатора на нужный под нашу задачу\nmodel.classifier[1] = torch.nn.Linear(1280, 41)\nmodel.to(device)\n# Подбираем оптимизатор и критерий останова под задачу классификации\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)\ncriterion = torch.nn.CrossEntropyLoss()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.782718Z","iopub.execute_input":"2024-10-21T17:56:52.783103Z","iopub.status.idle":"2024-10-21T17:56:52.800901Z","shell.execute_reply.started":"2024-10-21T17:56:52.783063Z","shell.execute_reply":"2024-10-21T17:56:52.799597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\nimport librosa\nimport cv2\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.802744Z","iopub.execute_input":"2024-10-21T17:56:52.803093Z","iopub.status.idle":"2024-10-21T17:56:52.809173Z","shell.execute_reply.started":"2024-10-21T17:56:52.803055Z","shell.execute_reply":"2024-10-21T17:56:52.807909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = train_path\nTEST_PATH = '../input/freesound-audio-tagging/audio_test/'\nbatch_size = 64\nepochs = 10\n\nclass SoundDataset(Dataset):\n    def __init__(self, dataframe, path, test=False):\n        super(SoundDataset, self).__init__()\n        self.dataframe = dataframe\n        self.path = path\n        self.test = test\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        file_path = self.dataframe.fname.values[idx]\n        label = self.dataframe.label.values[idx]\n        path = (TEST_PATH if self.test else TRAIN_PATH) + file_path\n        signal, _ = librosa.load(path)\n        signal = librosa.feature.melspectrogram(y=signal)\n        signal = librosa.power_to_db(signal, ref=np.max)\n\n        try:\n            resized = cv2.resize(signal, (128, 128))\n        except Exception as e:\n            print(path)\n            print(str(e))\n            resized = np.zeros(shape=(128, 128))\n\n        X = np.stack([resized] * 3)  # Дублирование каналов\n        X = torch.tensor(X, dtype=torch.float32)\n\n        if not self.test:\n            y = label_encoder[label]\n            return X, y\n        else:\n            return X","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.811002Z","iopub.execute_input":"2024-10-21T17:56:52.811389Z","iopub.status.idle":"2024-10-21T17:56:52.824306Z","shell.execute_reply.started":"2024-10-21T17:56:52.811338Z","shell.execute_reply":"2024-10-21T17:56:52.823134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_validation, y_train, y_validation = train_test_split(train, train, test_size=0.2, shuffle=True, random_state=5)\ntrain_dataset = SoundDataset(x_train, TRAIN_PATH)\nval_dataset = SoundDataset(x_validation, TRAIN_PATH)\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.826100Z","iopub.execute_input":"2024-10-21T17:56:52.826470Z","iopub.status.idle":"2024-10-21T17:56:52.846565Z","shell.execute_reply.started":"2024-10-21T17:56:52.826430Z","shell.execute_reply":"2024-10-21T17:56:52.845007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in range(epochs):\n    model.train()\n    train_loss = 0\n    train_correct = 0\n    for X, y in train_loader:\n        X, y = X.to(device), y.to(device)\n        optimizer.zero_grad()\n        pred = model(X)\n        loss = criterion(pred, y)\n        loss.backward()\n        optimizer.step()\n        train_loss += loss.item() * X.size(0)\n        train_correct += torch.sum(pred.argmax(1) == y).item()\n\n    train_loss = train_loss / len(train_loader.dataset)\n    train_accuracy = train_correct / len(train_loader.dataset)\n\n    model.eval()\n    val_loss = 0\n    val_correct = 0\n    with torch.no_grad():\n        for X, y in val_loader:\n            X, y = X.to(device), y.to(device)\n            pred = model(X)\n            loss = criterion(pred, y)\n            val_loss += loss.item() * X.size(0)\n            val_correct += torch.sum(pred.argmax(1) == y).item()\n\n    val_loss = val_loss / len(val_loader.dataset)\n    val_accuracy = val_correct / len(val_loader.dataset)\n\n    print(\"Epoch {}, Train Loss: {:.4f}, Train Accuracy: {:.4f}, Val Loss: {:.4f}, Val Accuracy: {:.4f}\".format(epoch+1, train_loss, train_accuracy, val_loss, val_accuracy))","metadata":{"execution":{"iopub.status.busy":"2024-10-21T17:56:52.848207Z","iopub.execute_input":"2024-10-21T17:56:52.848569Z","iopub.status.idle":"2024-10-21T20:31:27.492107Z","shell.execute_reply.started":"2024-10-21T17:56:52.848530Z","shell.execute_reply":"2024-10-21T20:31:27.488223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/freesound-audio-tagging/sample_submission.csv')\ntest_dataset = SoundDataset(test, TEST_PATH, test=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\npredictions = torch.tensor([])\n\nmodel.eval()\nwith torch.no_grad():\n    for X in test_loader:\n        X = X.to(device)\n        y_hat = model(X)\n        predictions = torch.cat([predictions, y_hat.cpu()])\n\npredictions = F.softmax(predictions, dim=1).detach().numpy()\n\nsubmission_top1 = test.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T20:50:34.553526Z","iopub.execute_input":"2024-10-21T20:50:34.554345Z","iopub.status.idle":"2024-10-21T20:58:53.518790Z","shell.execute_reply.started":"2024-10-21T20:50:34.554296Z","shell.execute_reply":"2024-10-21T20:58:53.517418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(test)):\n    p = predictions[i, :]\n    idx = np.argmax(p)\n    submission_top1.label[i] = labels[idx]\n\nsubmission_top1.to_csv('submission_final.csv', index=False, header=True)\n\nsubmission_top1.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T20:41:21.041605Z","iopub.execute_input":"2024-10-21T20:41:21.042021Z","iopub.status.idle":"2024-10-21T20:41:22.501279Z","shell.execute_reply.started":"2024-10-21T20:41:21.041977Z","shell.execute_reply":"2024-10-21T20:41:22.500025Z"},"trusted":true},"execution_count":null,"outputs":[]}]}