{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport math\nimport os\nimport cv2\nimport IPython.display as ipd \nimport librosa \nimport librosa.display\nimport torch\nimport numpy as np\nimport torch.nn.functional as F\nimport torchvision\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data.dataset import Dataset\nfrom torch.utils.data import DataLoader\nfrom torchvision import transforms\nfrom torchvision.models import resnet101","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-19T17:53:43.453301Z","iopub.execute_input":"2023-12-19T17:53:43.453735Z","iopub.status.idle":"2023-12-19T17:53:49.527612Z","shell.execute_reply.started":"2023-12-19T17:53:43.453696Z","shell.execute_reply":"2023-12-19T17:53:49.526401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nprint(device)\ntrain_path = '../input/freesound-audio-tagging/audio_train/'\ntrain = pd.read_csv(\"../input/freesound-audio-tagging/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:49.529627Z","iopub.execute_input":"2023-12-19T17:53:49.530189Z","iopub.status.idle":"2023-12-19T17:53:49.565290Z","shell.execute_reply.started":"2023-12-19T17:53:49.530156Z","shell.execute_reply":"2023-12-19T17:53:49.564466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels = len(train.label.unique())\nlabels = np.unique(train.label.values)\nlabel_encoder = {label:i for i, label in enumerate(labels)}","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:49.566815Z","iopub.execute_input":"2023-12-19T17:53:49.567440Z","iopub.status.idle":"2023-12-19T17:53:49.595219Z","shell.execute_reply.started":"2023-12-19T17:53:49.567408Z","shell.execute_reply":"2023-12-19T17:53:49.594024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = (128, 128)\nTRAIN_PATH = '../input/freesound-audio-tagging/audio_train/'\nTEST_PATH = '../input/freesound-audio-tagging/audio_test/'\n\nclass Dataset(Dataset):\n    def __init__(self, dataframe, test=False):\n        self.dataframe = dataframe\n        self.test = test\n        \n    def __getitem__(self, index):\n\n        X = np.zeros(shape=(3,IMG_SIZE[0], IMG_SIZE[1]))\n        FILE = self.dataframe.fname.values[index]\n        LABEL = self.dataframe.label.values[index]\n        \n        path = (TEST_PATH if self.test else TRAIN_PATH) + FILE\n        signal, _ = librosa.load(path)\n        signal = librosa.feature.melspectrogram(y=signal)    \n        signal = librosa.power_to_db(signal, ref=np.max) \n        \n        try:\n            resized = cv2.resize(signal, (IMG_SIZE[1], IMG_SIZE[0]))\n        except Exception as e:\n            print(path)\n            print(str(e))\n            resized = np.zeros(shape=(IMG_SIZE[1], IMG_SIZE[0]))\n        \n        for j in range(3):\n                X[j,:,:] = resized\n\n        if self.test == False:\n            y = label_encoder[LABEL]\n            return torch.tensor(X, dtype=torch.float), y\n        else:\n             return torch.tensor(X, dtype=torch.float)\n        \n    def __len__(self):\n        return self.dataframe.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:49.598388Z","iopub.execute_input":"2023-12-19T17:53:49.598773Z","iopub.status.idle":"2023-12-19T17:53:49.611447Z","shell.execute_reply.started":"2023-12-19T17:53:49.598740Z","shell.execute_reply":"2023-12-19T17:53:49.610359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\n\nx_train, x_validation, y_train, y_validation = train_test_split(train, train, test_size=0.2, shuffle=True, random_state=5)\ntrain_set = Dataset(x_train)\nval_set = Dataset(x_validation)\ntrain_loader = DataLoader(train_set, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_set , batch_size=batch_size, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:49.612928Z","iopub.execute_input":"2023-12-19T17:53:49.613271Z","iopub.status.idle":"2023-12-19T17:53:49.633976Z","shell.execute_reply.started":"2023-12-19T17:53:49.613234Z","shell.execute_reply":"2023-12-19T17:53:49.633102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = resnet101(pretrained=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:49.635312Z","iopub.execute_input":"2023-12-19T17:53:49.636175Z","iopub.status.idle":"2023-12-19T17:53:51.880041Z","shell.execute_reply.started":"2023-12-19T17:53:49.636141Z","shell.execute_reply":"2023-12-19T17:53:51.879075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_ftrs = model.fc.in_features\nmodel.fc = torch.nn.Linear(num_ftrs, num_labels)\nmodel = model.to(device)\nmodel.to(device);","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:51.881717Z","iopub.execute_input":"2023-12-19T17:53:51.882894Z","iopub.status.idle":"2023-12-19T17:53:51.910119Z","shell.execute_reply.started":"2023-12-19T17:53:51.882850Z","shell.execute_reply":"2023-12-19T17:53:51.908813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 8\noptimizer = torch.optim.AdamW(model.parameters(), lr=0.001)\ncost = torch.nn.CrossEntropyLoss()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:51.911751Z","iopub.execute_input":"2023-12-19T17:53:51.912396Z","iopub.status.idle":"2023-12-19T17:53:51.923072Z","shell.execute_reply.started":"2023-12-19T17:53:51.912354Z","shell.execute_reply":"2023-12-19T17:53:51.921269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in range(epochs):\n    train_loss = 0\n    val_loss = 0\n    train_correct = 0\n    val_correct = 0\n    model.train()\n    for x, y in train_loader:\n        optimizer.zero_grad()\n        x,y = x.to(device),y.to(device)\n        pred = model(x)\n        loss = cost(pred, y)\n        train_loss += cost(pred, y).item()\n        train_correct += (pred.argmax(1) == y).type(torch.float).sum().item()\n        loss.backward()\n        optimizer.step()\n\n    model.eval()\n    with torch.no_grad():\n        for x, y in val_loader:\n            x,y = x.to(device),y.to(device)\n            pred = model(x)\n            loss = cost(pred, y)\n            val_loss += cost(pred, y).item()\n            val_correct += (pred.argmax(1) == y).type(torch.float).sum().item()\n    train_loss = train_loss/len(train_loader)\n    val_loss = val_loss/len(val_loader)\n    train_accuracy = train_correct / len(x_train)\n    val_accuracy = val_correct / len(x_validation)\n    print(\"epoch = %d, train_loss = %.5f, val_loss = %.5f, train_accuracy = %.5f, val_accuracy = %.5f\" % (epoch, train_loss, val_loss, train_accuracy, val_accuracy))","metadata":{"execution":{"iopub.status.busy":"2023-12-19T17:53:51.924529Z","iopub.execute_input":"2023-12-19T17:53:51.925171Z","iopub.status.idle":"2023-12-19T22:22:13.183121Z","shell.execute_reply.started":"2023-12-19T17:53:51.925130Z","shell.execute_reply":"2023-12-19T22:22:13.176685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/freesound-audio-tagging/sample_submission.csv')\n\ntest_dataset = Dataset(test, test=True)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\npredictions = torch.tensor([])\nmodel.eval()\nfor x in test_loader:\n    x = x.to(device)\n    with torch.no_grad():\n        y_hat = model(x)\n    predictions = torch.cat([predictions, y_hat.cpu()])","metadata":{"execution":{"iopub.status.busy":"2023-12-19T22:22:13.197229Z","iopub.execute_input":"2023-12-19T22:22:13.201008Z","iopub.status.idle":"2023-12-19T22:39:54.937335Z","shell.execute_reply.started":"2023-12-19T22:22:13.200953Z","shell.execute_reply":"2023-12-19T22:39:54.936048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = F.softmax(predictions, dim=1).detach().numpy()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T22:39:54.939124Z","iopub.execute_input":"2023-12-19T22:39:54.939506Z","iopub.status.idle":"2023-12-19T22:39:54.955281Z","shell.execute_reply.started":"2023-12-19T22:39:54.939472Z","shell.execute_reply":"2023-12-19T22:39:54.954266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_top1 = test.copy()\n\nN = len(test)\nfor i in range(N):\n    p = predictions[i, :]\n    idx = np.argmax(p)\n    submission_top1.label[i] = labels[idx]\n\nsubmission_top1.to_csv('submission.csv', index=False, header=True)\n\nsubmission_top1.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-19T22:39:54.957310Z","iopub.execute_input":"2023-12-19T22:39:54.958132Z","iopub.status.idle":"2023-12-19T22:39:56.309513Z","shell.execute_reply.started":"2023-12-19T22:39:54.958096Z","shell.execute_reply":"2023-12-19T22:39:56.308186Z"},"trusted":true},"execution_count":null,"outputs":[]}]}