{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdClef2023 Mel torch CNN\nhttps://www.kaggle.com/code/stpeteishii/birdclef2023-sound-j-0-to-mel/notebook","metadata":{"papermill":{"duration":0.006545,"end_time":"2023-04-07T14:31:40.434181","exception":false,"start_time":"2023-04-07T14:31:40.427636","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport cv2\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data.sampler import SubsetRandomSampler\nfrom torch.utils.data import Dataset\nfrom torchvision import datasets, transforms, models \nfrom torchvision.utils import make_grid\nfrom torchvision.datasets import ImageFolder","metadata":{"papermill":{"duration":2.855174,"end_time":"2023-04-07T14:31:43.294813","exception":false,"start_time":"2023-04-07T14:31:40.439639","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-11T04:02:11.255465Z","iopub.execute_input":"2023-04-11T04:02:11.255987Z","iopub.status.idle":"2023-04-11T04:02:11.268681Z","shell.execute_reply.started":"2023-04-11T04:02:11.255939Z","shell.execute_reply":"2023-04-11T04:02:11.267167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set Train and Test","metadata":{"papermill":{"duration":0.005059,"end_time":"2023-04-07T14:31:43.607838","exception":false,"start_time":"2023-04-07T14:31:43.602779","status":"completed"},"tags":[]}},{"cell_type":"code","source":"paths=[]\nlabels=[]\nfor dirname, _, filenames in os.walk('/kaggle/input/birdclef2023-sound-j-0-to-mel'):\n    for filename in filenames:\n        if filename[-4:]=='.png' and 'results' not in filename:\n            path=os.path.join(dirname, filename)\n            paths+=[path]\n            label=path.split('/')[-1].split('_')[0]\n            labels+=[label]","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.278618Z","iopub.execute_input":"2023-04-11T04:02:11.279577Z","iopub.status.idle":"2023-04-11T04:02:11.300928Z","shell.execute_reply.started":"2023-04-11T04:02:11.279512Z","shell.execute_reply":"2023-04-11T04:02:11.299252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame(columns=['path','label'])\ndf['path']=paths\ndf['label']=labels\ndisplay(df['label'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.305266Z","iopub.execute_input":"2023-04-11T04:02:11.305736Z","iopub.status.idle":"2023-04-11T04:02:11.320560Z","shell.execute_reply.started":"2023-04-11T04:02:11.305689Z","shell.execute_reply":"2023-04-11T04:02:11.319078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df[df['label']!='crefra2']","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.323192Z","iopub.execute_input":"2023-04-11T04:02:11.323709Z","iopub.status.idle":"2023-04-11T04:02:11.331418Z","shell.execute_reply.started":"2023-04-11T04:02:11.323659Z","shell.execute_reply":"2023-04-11T04:02:11.330370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names=sorted(df['label'].unique().tolist())\nN=list(range(len(class_names)))\nprint(len(class_names))\nnormal_mapping=dict(zip(class_names,N)) \nreverse_mapping=dict(zip(N,class_names))\ndf['labeli']=df['label'].map(normal_mapping)\nprint(normal_mapping)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.332729Z","iopub.execute_input":"2023-04-11T04:02:11.333726Z","iopub.status.idle":"2023-04-11T04:02:11.346924Z","shell.execute_reply.started":"2023-04-11T04:02:11.333680Z","shell.execute_reply":"2023-04-11T04:02:11.345562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_transform=transforms.Compose([ \n    transforms.Resize(217),\n    transforms.ToTensor()])","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.350883Z","iopub.execute_input":"2023-04-11T04:02:11.351768Z","iopub.status.idle":"2023-04-11T04:02:11.357610Z","shell.execute_reply.started":"2023-04-11T04:02:11.351710Z","shell.execute_reply":"2023-04-11T04:02:11.356237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CAUTION!!! \npath0='/kaggle/input/birdclef2023-sound-j-0-to-mel/mel/brrwhe3_XC289781.png'\nimg = Image.open(path0)\nprint(type(img))\nimg1 = np.array(img)\nprint(type(img1),img1.shape)#(217, 217, 4)\nimg2=cv2.imread(path0)\nprint(type(img2),img2.shape)#(217, 217, 3)\n#display(train_transform(img))##OK\n#display(train_transform(img1))##impossible\n#display(train_transform(img2))##impossible","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.358975Z","iopub.execute_input":"2023-04-11T04:02:11.359374Z","iopub.status.idle":"2023-04-11T04:02:11.388102Z","shell.execute_reply.started":"2023-04-11T04:02:11.359335Z","shell.execute_reply":"2023-04-11T04:02:11.386719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        img_path = self.df.iloc[index]['path']\n        img = Image.open(img_path)        \n        if self.transform:\n            img = self.transform(img)\n        label = self.df.iloc[index]['labeli']\n\n        return img, label","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.389892Z","iopub.execute_input":"2023-04-11T04:02:11.390701Z","iopub.status.idle":"2023-04-11T04:02:11.400028Z","shell.execute_reply.started":"2023-04-11T04:02:11.390631Z","shell.execute_reply":"2023-04-11T04:02:11.398780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = CustomDataset(df, transform=train_transform)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.401947Z","iopub.execute_input":"2023-04-11T04:02:11.402914Z","iopub.status.idle":"2023-04-11T04:02:11.410833Z","shell.execute_reply.started":"2023-04-11T04:02:11.402857Z","shell.execute_reply":"2023-04-11T04:02:11.409720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#one by one\nfor images, labels in dataset:\n    print(type(images))\n    print(images.shape)#torch.Size([32, 4, 217, 217])\n    break","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:11.412617Z","iopub.execute_input":"2023-04-11T04:02:11.413139Z","iopub.status.idle":"2023-04-11T04:02:11.431776Z","shell.execute_reply.started":"2023-04-11T04:02:11.413086Z","shell.execute_reply":"2023-04-11T04:02:11.430051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_size = len(dataset)\nindices = list(range(dataset_size))\n\nsplit_ratio = 0.8\nsplit_index = int(dataset_size * split_ratio)\ntrain_indices = indices[:split_index]\ntest_indices = indices[split_index:]\n\ntrain_sampler = SubsetRandomSampler(train_indices)\ntest_sampler = SubsetRandomSampler(test_indices)\n\ntrain_loader = DataLoader(dataset, batch_size=32, sampler=train_sampler)\ntest_loader = DataLoader(dataset, batch_size=32, sampler=test_sampler)","metadata":{"papermill":{"duration":0.014007,"end_time":"2023-04-07T14:31:43.704637","exception":false,"start_time":"2023-04-07T14:31:43.690630","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-11T04:02:11.435437Z","iopub.execute_input":"2023-04-11T04:02:11.436906Z","iopub.status.idle":"2023-04-11T04:02:11.447042Z","shell.execute_reply.started":"2023-04-11T04:02:11.436810Z","shell.execute_reply":"2023-04-11T04:02:11.445548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Images and Labels","metadata":{"papermill":{"duration":0.004987,"end_time":"2023-04-07T14:31:43.716739","exception":false,"start_time":"2023-04-07T14:31:43.711752","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#batch by batch\nfor images, labels in train_loader:\n    print(type(images))    \n    print(images.shape)#torch.Size([32, 4, 217, 217])\n    break\nim=make_grid(images,nrow=8)","metadata":{"papermill":{"duration":0.245124,"end_time":"2023-04-07T14:31:43.966952","exception":false,"start_time":"2023-04-07T14:31:43.721828","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-11T04:02:11.449101Z","iopub.execute_input":"2023-04-11T04:02:11.449932Z","iopub.status.idle":"2023-04-11T04:02:11.930265Z","shell.execute_reply.started":"2023-04-11T04:02:11.449878Z","shell.execute_reply":"2023-04-11T04:02:11.928850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nplt.imshow(np.transpose(im.numpy(),(1,2,0)))\nplt.show()","metadata":{"papermill":{"duration":0.729679,"end_time":"2023-04-07T14:31:44.702005","exception":false,"start_time":"2023-04-07T14:31:43.972326","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-11T04:02:11.932180Z","iopub.execute_input":"2023-04-11T04:02:11.932601Z","iopub.status.idle":"2023-04-11T04:02:12.717987Z","shell.execute_reply.started":"2023-04-11T04:02:11.932557Z","shell.execute_reply":"2023-04-11T04:02:12.716820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN Model","metadata":{"papermill":{"duration":0.020431,"end_time":"2023-04-07T14:31:45.565962","exception":false,"start_time":"2023-04-07T14:31:45.545531","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class MyCNN(nn.Module):\n    \n    def __init__(self):\n        super(MyCNN, self).__init__()\n        self.conv1 = nn.Conv2d(4, 32, 3, padding=1)\n        self.relu1 = nn.ReLU(inplace=True)\n        self.conv2 = nn.Conv2d(32, 64, 3, padding=1)\n        self.relu2 = nn.ReLU(inplace=True)\n        self.maxpool = nn.MaxPool2d(2)\n        self.fc1 = nn.Linear(64 * 108 * 108, 128)  # 217 / 2 = 108\n        self.relu3 = nn.ReLU(inplace=True)\n        self.fc2 = nn.Linear(128, 16)\n\n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.relu1(x)\n        x = self.conv2(x)\n        x = self.relu2(x)\n        x = self.maxpool(x)\n        x = x.view(x.size(0), -1)\n        x = self.fc1(x)\n        x = self.relu3(x)\n        x = self.fc2(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:12.719327Z","iopub.execute_input":"2023-04-11T04:02:12.720407Z","iopub.status.idle":"2023-04-11T04:02:12.729549Z","shell.execute_reply.started":"2023-04-11T04:02:12.720365Z","shell.execute_reply":"2023-04-11T04:02:12.728428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=MyCNN()\ncriterion=nn.CrossEntropyLoss()\noptimizer=torch.optim.Adam(model.parameters(),lr=0.0001)","metadata":{"papermill":{"duration":0.391435,"end_time":"2023-04-07T14:31:46.068618","exception":false,"start_time":"2023-04-07T14:31:45.677183","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-11T04:02:12.732558Z","iopub.execute_input":"2023-04-11T04:02:12.733010Z","iopub.status.idle":"2023-04-11T04:02:13.789313Z","shell.execute_reply.started":"2023-04-11T04:02:12.732968Z","shell.execute_reply":"2023-04-11T04:02:13.787136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model","metadata":{"papermill":{"duration":0.030829,"end_time":"2023-04-07T14:31:46.120520","exception":false,"start_time":"2023-04-07T14:31:46.089691","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-11T04:02:13.791286Z","iopub.execute_input":"2023-04-11T04:02:13.792315Z","iopub.status.idle":"2023-04-11T04:02:13.801950Z","shell.execute_reply.started":"2023-04-11T04:02:13.792255Z","shell.execute_reply":"2023-04-11T04:02:13.800451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit","metadata":{"papermill":{"duration":0.021522,"end_time":"2023-04-07T14:31:46.163114","exception":false,"start_time":"2023-04-07T14:31:46.141592","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import time\nstart_time=time.time()\ntrain_losses=[]\ntest_losses=[]\ntrain_correct=[]\ntest_correct=[]\nepochs=30\n\nfor i in range(epochs):\n    #print(i)\n    trn_corr = 0\n    tst_corr = 0\n    for b, (X_train, y_train) in enumerate(train_loader):\n        b += 1\n        y_pred = model(X_train)\n        loss = criterion(y_pred, y_train)\n\n        predicted = torch.max(y_pred.data, 1)[1]\n        batch_corr = (predicted == y_train).sum()\n        trn_corr += batch_corr\n\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        if b % 200 == 0:\n            print(f\"epoch: {i} loss: {loss.item()} batch: {b} accuracy: {trn_corr.item() * 100 / (10 * b):7.3f}%\")\n    loss = loss.detach().numpy()\n    train_losses.append(loss)\n    train_correct.append(trn_corr)\n\n    with torch.no_grad():\n        for b, (X_test, y_test) in enumerate(test_loader):\n            y_val = model(X_test)\n            loss = criterion(y_val, y_test)\n\n            predicted = torch.max(y_val.data, 1)[1]\n            batch_corr = (predicted == y_test).sum()\n            tst_corr += batch_corr\n\n        loss = loss.detach().numpy()\n        test_losses.append(loss)\n        test_correct.append(tst_corr)\n\nprint(f'\\nDuration: {time.time() - start_time:.0f} seconds')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-11T04:02:13.804120Z","iopub.execute_input":"2023-04-11T04:02:13.805497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(train_losses,label=\"train_losses\")\nplt.plot(test_losses,label=\"test_losses\")\nplt.legend()\nplt.show()","metadata":{"papermill":{"duration":0.20267,"end_time":"2023-04-07T15:14:32.907497","exception":false,"start_time":"2023-04-07T15:14:32.704827","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save model\ntorch.save(model.state_dict(), 'model.pt') \n#new_model = ConvolutionalNetwork()\n#new_model.load_state_dict(torch.load('model.pt'))","metadata":{"papermill":{"duration":0.312005,"end_time":"2023-04-07T15:14:33.245183","exception":false,"start_time":"2023-04-07T15:14:32.933178","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict","metadata":{"papermill":{"duration":0.025518,"end_time":"2023-04-07T15:14:33.296690","exception":false,"start_time":"2023-04-07T15:14:33.271172","status":"completed"},"tags":[]}},{"cell_type":"code","source":"device = torch.device(\"cpu\")   #\"cuda:0\"\nmodel.eval()\n\ny_true=[]\ny_pred=[]\nwith torch.no_grad():\n    for test_data in test_loader:\n        test_images, test_labels = test_data[0].to(device), test_data[1].to(device)\n        pred = model(test_images).argmax(dim=1)\n        for i in range(len(pred)):\n            y_true.append(test_labels[i].item())\n            y_pred.append(pred[i].item())\nprint(y_pred[0:5])","metadata":{"papermill":{"duration":5.755721,"end_time":"2023-04-07T15:14:39.078938","exception":false,"start_time":"2023-04-07T15:14:33.323217","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_true,y_pred,target_names=class_names,digits=4))","metadata":{"papermill":{"duration":0.041117,"end_time":"2023-04-07T15:14:39.146519","exception":false,"start_time":"2023-04-07T15:14:39.105402","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.026004,"end_time":"2023-04-07T15:14:39.198628","exception":false,"start_time":"2023-04-07T15:14:39.172624","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.026349,"end_time":"2023-04-07T15:14:39.251399","exception":false,"start_time":"2023-04-07T15:14:39.225050","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}