{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom sklearn.model_selection import train_test_split\nimport json\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nin_dir = '/kaggle/input/cassava-leaf-disease-classification'\ntraining_files = os.path.join(in_dir, 'train_images')\ntraining_csv = os.path.join(in_dir, 'train.csv')\nlabel_mapper = os.path.join(in_dir, 'label_num_to_disease_map.json')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir(in_dir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(training_csv)\n#Get name mapper\nwith open(label_mapper, 'r') as f:\n    mapper = json.load(f)\nmapper\n\ndf['class_name'] = df.label.astype(str).map(mapper)\n\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train, df_valid = train_test_split(df, test_size = 0.2)\nprint(df_train)\nprint(df_valid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from torch.utils.data.sampler import SubsetRandomSampler\n\ntrain_indices = list(df_train.index)\nvalid_indices = list(df_valid.index)\n\n# Creating PT data samplers and loaders:\ntrain_sampler = SubsetRandomSampler(train_indices)\nvalid_sampler = SubsetRandomSampler(valid_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torchvision.transforms as transforms\n# https://discuss.pytorch.org/t/understanding-transform-normalize/21730\n# Normalizing image channels within range [-1,1], image = (image - mean) / std \ntransform = transforms.Compose(\n    [transforms.Resize((150,150)),\n     transforms.RandomHorizontalFlip(),\n     transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Custom dataset example from https://github.com/utkuozbulak/pytorch-custom-dataset-examples\nfrom torch.utils.data import Dataset\nfrom PIL import Image\n\nclass Cassava_Dataset(Dataset):\n    def __init__(self, img_data,img_path,transform=None):\n        self.img_path = img_path\n        self.transform = transform\n        self.img_data = img_data\n        \n    def __len__(self):\n        return len(self.img_data)\n    \n    def __getitem__(self, index):\n        img_name = os.path.join(self.img_path,self.img_data.loc[index, 'image_id'])\n        image = Image.open(img_name)\n        label = torch.tensor(self.img_data.loc[index, 'label'])\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainPath = '../input/cassava-leaf-disease-classification/train_images'\ndataset = Cassava_Dataset(df, trainPath, transform)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\n\nbatch_size = 128\n\ntrain_loader = torch.utils.data.DataLoader(dataset, batch_size=batch_size, \n                                           sampler=train_sampler)\nvalidation_loader = torch.utils.data.DataLoader(dataset, batch_size=batch_size,\n                                                sampler=valid_sampler)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def img_display(img):\n    img = img / 2 + 0.5     # unnormalize\n    npimg = img.numpy()\n    npimg = np.transpose(npimg, (1, 2, 0))\n    return npimg","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# get some random training images\ndataiter = iter(train_loader)\nimages, labels = dataiter.next()\narthopod_types = {0: 'Cassava Bacterial Blight (CBB)', 1: 'Cassava Brown Streak Disease (CBSD)', 2: 'Cassava Green Mottle (CGM)', 3:'Cassava Mosaic Disease (CMD)', 4: 'Healthy'}\n# Viewing data examples used for training\nfig, axis = plt.subplots(3, 4, figsize=(25, 10))\nfor i, ax in enumerate(axis.flat):\n    with torch.no_grad():\n        image, label = images[i], labels[i]\n        ax.imshow(img_display(image)) # add image\n        ax.set(title = f\"{arthopod_types[label.item()]}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch.nn as nn\n\nclass Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        #Output size after convolution filter\n        #((w-f+2P)/s) +1\n        \n        #Input shape= (256,3,150,150)\n        \n        self.conv1=nn.Conv2d(in_channels=3,out_channels=12,kernel_size=3,stride=1,padding=1)\n        #Shape= (256,12,150,150)\n        self.bn1=nn.BatchNorm2d(num_features=12)\n        #Shape= (256,12,150,150)\n        self.relu1=nn.ReLU()\n        #Shape= (256,12,150,150)\n        \n        self.pool=nn.MaxPool2d(kernel_size=2)\n        #Reduce the image size be factor 2\n        #Shape= (256,12,75,75)\n        \n        \n        self.conv2=nn.Conv2d(in_channels=12,out_channels=20,kernel_size=3,stride=1,padding=1)\n        #Shape= (256,20,75,75)\n        self.relu2=nn.ReLU()\n        #Shape= (256,20,75,75)\n        \n        \n        \n        self.conv3=nn.Conv2d(in_channels=20,out_channels=32,kernel_size=3,stride=1,padding=1)\n        #Shape= (256,32,75,75)\n        self.bn3=nn.BatchNorm2d(num_features=32)\n        #Shape= (256,32,75,75)\n        self.relu3=nn.ReLU()\n        #Shape= (256,32,75,75)\n        \n        \n        self.fc=nn.Linear(in_features=75 * 75 * 32,out_features=5)\n        \n    def forward(self, x):\n        output=self.conv1(x)\n        output=self.bn1(output)\n        output=self.relu1(output)\n            \n        output=self.pool(output)\n            \n        output=self.conv2(output)\n        output=self.relu2(output)\n            \n        output=self.conv3(output)\n        output=self.bn3(output)\n        output=self.relu3(output)\n            \n            \n            #Above output will be in matrix form, with shape (256,32,75,75)\n            \n        output=output.view(-1,32*75*75)\n            \n            \n        output=self.fc(output)\n            \n        return output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n\n# Assuming that we are on a CUDA machine, this should print a CUDA device:\n\nprint(device)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model = Net() # On CPU\nmodel = Net().to(device)  # On GPU\nprint(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch.optim as optim\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001, weight_decay=0.0001)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def accuracy(out, labels):\n    _,pred = torch.max(out, dim=1)\n    return torch.sum(pred==labels).item()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch.nn.functional as F\n\nn_epochs = 24\nprint_every = 10\nvalid_loss_min = np.Inf\nval_loss = []\nval_acc = []\ntrain_loss = []\ntrain_acc = []\ntotal_step = len(train_loader)\nfor epoch in range(1, n_epochs+1):\n    running_loss = 0.0\n    # scheduler.step(epoch)\n    correct = 0\n    total=0\n    print(f'Epoch {epoch}\\n')\n    for batch_idx, (data_, target_) in enumerate(train_loader):\n        data_, target_ = data_.to(device), target_.to(device)# on GPU\n        # zero the parameter gradients\n        optimizer.zero_grad()\n        # forward + backward + optimize\n        outputs = model(data_)\n        loss = criterion(outputs, target_)\n        loss.backward()\n        optimizer.step()\n        # print statistics\n        running_loss += loss.item()\n        _,pred = torch.max(outputs, dim=1)\n        correct += torch.sum(pred==target_).item()\n        total += target_.size(0)\n        if (batch_idx) % 20 == 0:\n            print ('Epoch [{}/{}], Step [{}/{}], Loss: {:.4f}' \n                   .format(epoch, n_epochs, batch_idx, total_step, loss.item()))\n    train_acc.append(100 * correct / total)\n    train_loss.append(running_loss/total_step)\n    print(f'\\ntrain loss: {np.mean(train_loss):.4f}, train acc: {(100 * correct / total):.4f}')\n    batch_loss = 0\n    total_t=0\n    correct_t=0\n    with torch.no_grad():\n        model.eval()\n        for data_t, target_t in (validation_loader):\n            data_t, target_t = data_t.to(device), target_t.to(device)# on GPU\n            outputs_t = model(data_t)\n            loss_t = criterion(outputs_t, target_t)\n            batch_loss += loss_t.item()\n            _,pred_t = torch.max(outputs_t, dim=1)\n            correct_t += torch.sum(pred_t==target_t).item()\n            total_t += target_t.size(0)\n        val_acc.append(100 * correct_t / total_t)\n        val_loss.append(batch_loss/len(validation_loader))\n        network_learned = batch_loss < valid_loss_min\n        print(f'validation loss: {np.mean(val_loss):.4f}, validation acc: {(100 * correct_t / total_t):.4f}\\n')\n        # Saving the best weight \n        if network_learned:\n            valid_loss_min = batch_loss\n            torch.save(model.state_dict(), 'model_classification_tutorial.pt')\n            print('Detected network improvement, saving current model')\n    model.train()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}