{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"scrolled":true},"cell_type":"code","source":"\n# Start up  basic Libraries\nimport os\nimport pandas as pd\npath = '../input/cassava-leaf-disease-classification'\nos.listdir(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"#  Read data\nmarco = pd.read_csv(path + \"/train.csv\")\nmarco.head()\n\n# Path to each image for processing\nmarco[\"path\"] = marco['image_id'].map(lambda x: path+\"/train_images/\"+x)\nmarco = marco.drop(columns=['image_id'])\nmarco= marco.sample(frac=1).reset_index(drop=True)\n\n#marco.head()\n\n# Train set size can also make a significant difference in the model's behavior.\ntrain = int(len(marco) * 0.7)\nval = len(marco) - train\ntrain_marco = marco[:train]\nval_marco = marco[train + 1:].reset_index().drop(columns = ['index'])\n\ntrain_marco.head()\nval_marco.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"from PIL import Image  #Image manipulator and see data\nimagexample = Image.open(train_marco['path'][1]) # or any other from set\nimagexample","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"# Libraries to build and execute de cnn\nimport torch\nimport torch.nn.functional as fnn\nimport torch.nn as nn\nimport torchvision\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset, DataLoader\nimport matplotlib.image as img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"class Leafdisease(Dataset):\n\n    def __init__(self, dataframe, transform = None):\n        super().__init__()\n        self.marco = dataframe\n        self.transform = transform\n        \n        \n    def __len__(self):\n        return len(self.marco[\"path\"])\n    \n    def __getitem__(self, index):\n        path = self.marco[\"path\"][index]\n        label = self.marco[\"label\"][index]\n        with open(path, 'rb') as fnn:\n            image = Image.open(fnn)\n            image = image.convert(\"RGB\")\n            \n        if self.transform is not None:\n            image = self.transform(image)\n        \n        return image, label\n\nDataset = Leafdisease(train_marco)\nimage, label = Dataset.__getitem__(0)\ntrain_transform = transforms.Compose([transforms.Resize((256, 256)), \n                        transforms.RandomHorizontalFlip(),transforms.RandomRotation(degrees = 45),\n                         transforms.ToTensor(), transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                          std=[0.229, 0.224, 0.225])])\n\nval_transform = transforms.Compose([transforms.Resize((256, 256)),\n                                      transforms.ToTensor(),\n                                      transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"# dataset for training and validating in nn\ntrain_data = Leafdisease(train_marco, train_transform)\nvalid_data = Leafdisease(val_marco, val_transform)\n\nepoch = 10\nnum_classes = 5\nbatch_size = 32\nlr = 0.1\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else \"cpu\")\n\ntrain_loader = DataLoader(dataset = train_data, \n                          batch_size = batch_size,\n                          shuffle=True,\n                          num_workers = 0)\n\nvalid_loader = DataLoader(dataset = valid_data, \n                          batch_size = batch_size,\n                          shuffle=False,\n                          num_workers = 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"mean=[0.485, 0.456, 0.406]\nstd=[0.229, 0.224, 0.225]\n\nimage, label = next(iter(train_loader))\nimage = image[0]\nimage = image * torch.tensor(std).view(3, 1, 1)\nimage = image + torch.tensor(mean).view(3, 1, 1)\nimage = transforms.ToPILImage()(image)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"class ConvBlock(nn.Module):\n    def __init__(self, in_channels, out_channels):\n        super().__init__()\n        \n        self.conv1 = nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, 3, 1, 1),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(),\n        )\n        \n        self.conv2 = nn.Sequential(\n            nn.Conv2d(out_channels, out_channels, 3, 1, 1),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(),\n        )\n        self._init_weights()\n        \n    def _init_weights(self):\n        for m in self.modules():\n            if isinstance(m, nn.Conv2d):\n                nn.init.kaiming_normal_(m.weight)\n                if m.bias is not None:\n                    nn.init.zeros_(m.bias)\n            elif isinstance(m, nn.BatchNorm2d):\n                nn.init.constant_(m.weight, 1)\n                nn.init.zeros_(m.bias)\n    \n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        x = fnn.avg_pool2d(x, 2)\n        return x\n\nclass CNN(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.conv = nn.Sequential(\n            ConvBlock(in_channels = 3, out_channels = 64),\n            ConvBlock(in_channels = 64, out_channels = 128),\n            ConvBlock(in_channels = 128, out_channels = 256),\n            ConvBlock(in_channels = 256, out_channels = 512),\n        )\n        self.fc = nn.Sequential(\n            nn.Linear(512, 128),\n            nn.PReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(128, 64),\n            nn.PReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(64, 5),\n        )\n        \n    def forward(self, x):\n        x = self.conv(x)\n        x = torch.mean(x, dim = 3)\n        x, _ = torch.max(x, dim = 2)\n        # print(x.size())\n        x = self.fc(x)\n        \n        return x\n        \n        \nmodel = CNN().to(device)\n\nmodel\n\ncriterion = nn.CrossEntropyLoss()\noptim = torch.optim.Adam(model.parameters(), lr = lr)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# for data, target in train_loader:, also try fastai in R\n#    print(data.size(), target)\n\noutput_list = os.listdir(\"./\")\nbest_model = None\nbest_loss = float(\"inf\")\nif \"model.pth\" not in output_list:\n    train_losses = []\n    valid_losses = []\n    for epoch in range(1, epoch + 1):\n        train_loss = 0\n        valid_loss = 0\n\n        model.train()\n        for data, target in train_loader:\n            data = data.to(device)\n            # print(data.size())\n            target = target.to(device)\n            optim.zero_grad()\n            # print(data.size())\n            output = model(data)\n            loss = criterion(output, target)\n            loss.backward()\n            optim.step()\n            train_loss += loss.item()*len(data)\n\n        train_loss = train_loss/len(train_loader.sampler)\n        train_losses.append(train_loss)\n\n        model.eval()\n        for data, target in valid_loader:\n            data = data.to(device)\n            target = target.to(device)\n\n            output = model(data)\n            loss = criterion(output, target)\n\n            valid_loss += loss.item()*len(data)\n            if valid_loss < best_loss:\n                best_model = model\n                best_loss = valid_loss\n\n        valid_loss = valid_loss/len(valid_loader.sampler)\n        valid_losses.append(valid_loss)\n        print('Epoch: {} \\tTraining Loss: {:.5f} \\tValidation Loss: {:.5f}'.format(epoch, train_loss, valid_loss))\n\n    torch.save(best_model.state_dict(), \"./model.pth\")\n\n ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"# Calculating loss\nfrom matplotlib import pyplot as plt\n\ny = list(range(epoch))\ntrain_loss = plt.plot(y, train_losses)\nvalid_loss = plt.plot(y, valid_losses)\nplt.ylabel(\"loss\")\nplt.legend((train_loss[0], valid_loss[0]), (\"train loss\", \"valid loss\"),)\nplt.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.state_dict(torch.load(\"./model.pth\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true},"cell_type":"code","source":"import os\npath = \"../input/cassava-leaf-disease-classification/test_images/\"\nfrom PIL import Image\nos.listdir(path)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = []\ntest_image = os.listdir(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_path = []\nimage_id = []\nfor i in os.listdir(path):\n    image_id.append(str(i))\n    image_path.append(path + str(i))\n\npred = []\nfor path in image_path:    \n    image = Image.open(path)\n    image = val_transform(image)\n    image = image.unsqueeze(0).to(device)\n    predict = model(image).argmax(1).item()\n    pred.append(predict)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred\n\nFIN = pd.DataFrame({'image_id': image_id, 'label': pred})\n\nFIN\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"FIN.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}