{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport os\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision\nimport torchvision.transforms as transforms\nimport matplotlib.pyplot as plt\n\nfrom torch.utils.data import Dataset, DataLoader\n\nimport json\n\nfrom PIL import Image\n\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-10T18:53:45.109778Z","iopub.execute_input":"2021-07-10T18:53:45.110406Z","iopub.status.idle":"2021-07-10T18:53:45.648724Z","shell.execute_reply.started":"2021-07-10T18:53:45.110205Z","shell.execute_reply":"2021-07-10T18:53:45.647622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.652572Z","iopub.execute_input":"2021-07-10T18:53:45.652875Z","iopub.status.idle":"2021-07-10T18:53:45.679362Z","shell.execute_reply.started":"2021-07-10T18:53:45.652844Z","shell.execute_reply":"2021-07-10T18:53:45.678056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 3\nBATCH = 4\nLR = 0.003","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.682117Z","iopub.execute_input":"2021-07-10T18:53:45.682917Z","iopub.status.idle":"2021-07-10T18:53:45.694485Z","shell.execute_reply.started":"2021-07-10T18:53:45.68287Z","shell.execute_reply":"2021-07-10T18:53:45.693022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = transforms.Compose(\n    [transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))])","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.697006Z","iopub.execute_input":"2021-07-10T18:53:45.6976Z","iopub.status.idle":"2021-07-10T18:53:45.708263Z","shell.execute_reply.started":"2021-07-10T18:53:45.697554Z","shell.execute_reply":"2021-07-10T18:53:45.707108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = '../input/cassava-leaf-disease-classification/train_images/'\nTEST_DIR = '../input/cassava-leaf-disease-classification/test_images/'\n\nlabels = json.load(open(\"../input/cassava-leaf-disease-classification/label_num_to_disease_map.json\"))\ntrain = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\nsample = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\n\nX_train, Y_train = train['image_id'].values, train['label'].values\n\nX_test = [name for name in (os.listdir(TEST_DIR))]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.712371Z","iopub.execute_input":"2021-07-10T18:53:45.712771Z","iopub.status.idle":"2021-07-10T18:53:45.75222Z","shell.execute_reply.started":"2021-07-10T18:53:45.71274Z","shell.execute_reply":"2021-07-10T18:53:45.751184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class GetData(Dataset):\n    def __init__(self, Dir, FNames, Labels, Transform):\n        self.dir = Dir\n        self.fnames = FNames\n        self.transform = Transform\n        self.lbs = Labels\n        \n    def __len__(self):\n        return len(self.fnames)\n\n    def __getitem__(self, index):\n        \n        x = Image.open(os.path.join(self.dir, self.fnames[index]))\n        x = x.resize((32,32))\n        #x = x.resize((52, 52))\n        #x = x.resize((104, 104))\n        \n        if \"train\" in self.dir:    \n            return self.transform(x), self.lbs[index]            \n        elif \"test\" in self.dir:            \n            return self.transform(x), self.fnames[index]","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.754344Z","iopub.execute_input":"2021-07-10T18:53:45.754649Z","iopub.status.idle":"2021-07-10T18:53:45.763385Z","shell.execute_reply.started":"2021-07-10T18:53:45.754622Z","shell.execute_reply":"2021-07-10T18:53:45.762069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset = GetData(TRAIN_DIR, X_train, Y_train, transform)\ntrainloader = DataLoader(trainset, batch_size=BATCH, shuffle=True, num_workers=4)\n\ntestset = GetData(TEST_DIR, X_test, None, transform)\ntestloader = DataLoader(testset, batch_size=1, shuffle=False, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.765159Z","iopub.execute_input":"2021-07-10T18:53:45.766026Z","iopub.status.idle":"2021-07-10T18:53:45.776219Z","shell.execute_reply.started":"2021-07-10T18:53:45.765979Z","shell.execute_reply":"2021-07-10T18:53:45.775133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ConvNet(nn.Module):\n    def __init__(self):\n        #torch.Size[(4, 3, 32, 32)] # batches, chnls, nxm of each img\n        \n        #print(self.conv1) torch.Size[(4, 6, 28, 28)] # in this case (n + 2*p - f)/s + 1 is equal to n-f+1 (cause of stride = 1 and padding = 0)\n        #print(self.pool) torch.Size[(4, 6, 14, 14)]\n        #print(self.conv2) torch.Size[(6, 16, 10, 10)] \n        #print(self.pool2) torch.Size[(6, 16, 5, 5)]\n        \n        super(ConvNet, self).__init__()\n        self.conv1 = nn.Conv2d(3, 6, 5)\n        self.pool = nn.MaxPool2d(2, 2) #kernel size = 2, stride = 2\n        self.conv2 = nn.Conv2d(6, 16, 5)\n        self.fc1 = nn.Linear(16 * 5 * 5, 120)\n        self.fc2 = nn.Linear(120, 84)\n        self.fc3 = nn.Linear(84, 5)\n\n    def forward(self, x):\n        # -> n, 3, 32, 32\n        x = self.pool(F.relu(self.conv1(x)))  # -> n, 6, 14, 14\n        x = self.pool(F.relu(self.conv2(x)))  # -> n, 16, 5, 5\n        x = x.view(-1, 16 * 5 * 5)            # -> n, 400\n        x = F.relu(self.fc1(x))               # -> n, 120\n        x = F.relu(self.fc2(x))               # -> n, 84\n        x = self.fc3(x)                       # -> n, 5 #output\n        return x","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.78027Z","iopub.execute_input":"2021-07-10T18:53:45.78086Z","iopub.status.idle":"2021-07-10T18:53:45.794153Z","shell.execute_reply.started":"2021-07-10T18:53:45.780817Z","shell.execute_reply":"2021-07-10T18:53:45.792696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ConvNet().to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.SGD(model.parameters(), lr=LR)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:45.812685Z","iopub.execute_input":"2021-07-10T18:53:45.813229Z","iopub.status.idle":"2021-07-10T18:53:47.928294Z","shell.execute_reply.started":"2021-07-10T18:53:45.813156Z","shell.execute_reply":"2021-07-10T18:53:47.927195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_acc = 0\nn_total_steps = len(trainloader)\nfor epoch in range(EPOCHS):\n    \n    running_loss = 0\n    correct = 0\n    total = 0\n    \n    for i, (images, labels) in enumerate(trainloader):\n        # origin shape: [4, 3, 32, 32] = 4, 3, 1024\n        # input_layer: 3 input channels, 6 output channels, 5 kernel size\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        # Backward and optimize\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n        _, predict = torch.max(outputs.data, 1)\n        correct += (predict == labels).sum().item()\n        total += labels.size(0)\n        \n        train_acc = correct / total\n\n    print('Epoch: %d | Loss: %.4f | Accuracy: %.4f'%(epoch + 1, loss.item(), train_acc))\n\nprint('Finished Training')\ntorch.save(model.state_dict(), \"cnn.pth\")","metadata":{"execution":{"iopub.status.busy":"2021-07-10T18:53:47.930934Z","iopub.execute_input":"2021-07-10T18:53:47.931272Z","iopub.status.idle":"2021-07-10T19:06:16.45212Z","shell.execute_reply.started":"2021-07-10T18:53:47.931244Z","shell.execute_reply":"2021-07-10T19:06:16.45076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_ls = []\n\nwith torch.no_grad():\n    model.eval()\n    for image, fname in testloader: \n        image = image.to(device)\n        \n        logits = model(image)        \n        ps = torch.exp(logits)        \n        _, top_class = ps.topk(1, dim=1)\n        \n        for pred in top_class:\n            s_ls.append([fname[0], pred.item()])","metadata":{"execution":{"iopub.status.busy":"2021-07-10T19:06:16.455697Z","iopub.execute_input":"2021-07-10T19:06:16.456014Z","iopub.status.idle":"2021-07-10T19:06:16.649993Z","shell.execute_reply.started":"2021-07-10T19:06:16.455969Z","shell.execute_reply":"2021-07-10T19:06:16.648802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame.from_records(s_ls, columns=['image_id', 'label'])\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T19:06:16.654198Z","iopub.execute_input":"2021-07-10T19:06:16.654515Z","iopub.status.idle":"2021-07-10T19:06:16.676522Z","shell.execute_reply.started":"2021-07-10T19:06:16.654467Z","shell.execute_reply":"2021-07-10T19:06:16.675076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T19:06:16.678536Z","iopub.execute_input":"2021-07-10T19:06:16.679147Z","iopub.status.idle":"2021-07-10T19:06:16.687892Z","shell.execute_reply.started":"2021-07-10T19:06:16.679105Z","shell.execute_reply":"2021-07-10T19:06:16.686629Z"},"trusted":true},"execution_count":null,"outputs":[]}]}