{"cells":[{"metadata":{"_uuid":"cc5a309ffc6122a773ecf24f587472530e1f0ce4"},"cell_type":"markdown","source":"# CNN with PyTorch for Beginners"},{"metadata":{"_uuid":"6e93ff952ed0c54ee9064c7909842a0539be39c1"},"cell_type":"markdown","source":"1. Load and normalizing the MNIST datasets\n2. Define a Convolution Neural Network\n3. Define a loss function\n4. Train the network on the training data\n5. Evaluate the model\n6. Prediction and submition"},{"metadata":{"trusted":true,"_uuid":"21056eac4b2cd5af9c74d77591570fdcb84a300f"},"cell_type":"code","source":"# data loading and presentation\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# PyTorch\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7364db7a108aaaa61b57524d9254fd06da839cb"},"cell_type":"markdown","source":"### 1. Load data"},{"metadata":{"trusted":true,"_uuid":"baa9532454a92797dc9a496d53cfa28326431cc3"},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"95d78f41162949897f2b62420ebfcdbd21d96976"},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71d2f0064bbf53dc615f86c28d9199e767d04e92"},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"eca6f1bf3353620fa49845434f1583031364f584"},"cell_type":"markdown","source":"Our training dataset has 42000 rows(i.e. 42000 samples), each row has a 'label' column and 784 'pixel' columns(represent a $28 \\times 28$ pixel picture). Let's see a few of them."},{"metadata":{"trusted":true,"_uuid":"834f3aaf04b75b39c0d5c1516dbc57fc95157c2d"},"cell_type":"code","source":"for i in range(8):\n    plt.subplot(1, 8, i+1)\n    plt.imshow(train_df.iloc[i, 1:].values.reshape(28, 28), cmap='gray')\n    print(train_df.iloc[i, 0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c36b4fad792d151660017c1dde5b6028a2fb7a0d"},"cell_type":"code","source":"train_df.iloc[i, 1:].plot(kind='hist')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7560158239a3d333bc053a607b55139719f9596e"},"cell_type":"markdown","source":"The value of each pixel is 0~255, we'd better normalize it(devided by 255). `torch.utils.data.Dataset` is a convenient tools for data loading provided by PyTorch.\n\nIt is an abstract class representing a dataset. Your custom dataset should inherit Dataset and override the following methods:\n\n- `__len__` so that len(dataset) returns the size of the dataset.\n- `__getitem__` to support the indexing such that dataset[i] can be used to get ith sample"},{"metadata":{"trusted":true,"_uuid":"3c122217e9407f3bab68fb36f172f639c0d42c39"},"cell_type":"code","source":"class MNIST_dataset(Dataset):\n    def __init__(self, df, rows=42000):\n        self.imgnp = df.iloc[:rows, 1:].values\n        self.labels = df.iloc[:rows, 0].values\n        self.rows = rows\n    \n    def __len__(self):\n        return self.rows\n    \n    def __getitem__(self, idx):\n        image = torch.tensor(self.imgnp[idx], dtype=torch.float) / 255  # Normalize\n        image = image.view(1, 28, 28)  # (channel, height, width)\n        label = self.labels[idx]\n        return (image, label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"223781b5c1232d40a0b96e81e0ebe8fafd45888d"},"cell_type":"code","source":"trainloader = DataLoader(MNIST_dataset(train_df, 42000), batch_size=4, shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f23baf36115ffd0cddb9c9b31f11ebff3374ec94"},"cell_type":"code","source":"dataiter = iter(trainloader)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7eae53fa21b41f2ce955bb55c74b746fc9182a1"},"cell_type":"code","source":"images, labels = dataiter.next()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5f2ac95c714aad01f8d2683bc4c0e625f269806"},"cell_type":"code","source":"images.size(), labels.size()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57a7a391fef80e1346fa98a4eecf8559b3cab840"},"cell_type":"code","source":"for i in range(4):\n    plt.subplot(1, 4, i+1)\n    plt.imshow(images[i, 0], cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19cf0abececaf87f33cef923e0f85ec4ad43c918"},"cell_type":"code","source":"labels","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b1bce9c4b41c4f624716ee883e65117524a74aaf"},"cell_type":"markdown","source":"## 2. Define a Convolution Neural Network"},{"metadata":{"_uuid":"9a1b5c717db6398897fa8b1dd13c0530a3ab7496"},"cell_type":"markdown","source":"Our CNN model is as follow\n![CNN model](https://pytorch.org/tutorials/_images/mnist.png)"},{"metadata":{"trusted":true,"_uuid":"876e2f4506cab849b2a357238ebd72a20374d932"},"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        self.conv1 = nn.Conv2d(1, 6, 5)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(6, 16, 5)\n        self.fc1 = nn.Linear(16 * 4 * 4, 120)\n        self.fc2 = nn.Linear(120, 84)\n        self.fc3 = nn.Linear(84, 10)\n        \n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.pool(F.relu(self.conv2(x)))\n        x = x.view(-1, 16 * 4 * 4)\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = self.fc3(x)\n        return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8893297ac76f1fe079e08040f14bc925a7f745e"},"cell_type":"code","source":"net = Net()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f7bddddff80296d4f470ccc8bb7fe32d2d1005c1"},"cell_type":"markdown","source":"## 3. Loss Function"},{"metadata":{"trusted":true,"_uuid":"58b87f5a111106c4f203eb25632aeeb11a1ff1b7"},"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\noptimizer = optim.SGD(net.parameters(), lr=0.01, momentum=0.9)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4b2bed1951c4cff7101f57b6b99c5105695a4cd5"},"cell_type":"markdown","source":"lr(learning rate) is a very important hyperparameter. Small lr will make the traning process very slow, but increase lr may also cause overshooting the global optimum and cause divergent. A good method is plot the Learning rate curve."},{"metadata":{"_uuid":"115bdc6198d6fcd1be0c38de217b1e2b73019b81"},"cell_type":"markdown","source":"## 4. Train the network on the training data"},{"metadata":{"scrolled":false,"trusted":true,"_uuid":"14f96f00aa8bf9586152891a3bd4f9e1e82d71cc"},"cell_type":"code","source":"running_loss_list = []\nfor epoch in range(2):\n    running_loss = 0.0\n    for i, data in enumerate(trainloader, 0):\n        inputs, labels = data\n        optimizer.zero_grad()\n        \n        outputs = net(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        # print statistics\n        running_loss += loss.item()\n        if i % 800 == 799:\n            print('[%d, %5d] loss: %.3f' %\n                 (epoch + 1, i + 1, running_loss / 800)\n                 )\n            running_loss_list.append(running_loss)\n            running_loss = 0.0\nprint('Finished Training')","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"9501950e78219b51f6de27324615bfd879e860c6"},"cell_type":"code","source":"plt.plot(running_loss_list)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"510872f3f6e8b3dcbb58b1151098c3a7f60115bc"},"cell_type":"markdown","source":"The Learning rate curve is divergent, which means the learning rate is to large, we should decrease it."},{"metadata":{"scrolled":false,"trusted":true,"_uuid":"4f7d3af190186c1df0c7fbfb9114f972b35bec23"},"cell_type":"code","source":"net = Net()\noptimizer = optim.SGD(net.parameters(), lr=0.001, momentum=0.9)\n\nrunning_loss_list = []\nfor epoch in range(2):\n    running_loss = 0.0\n    for i, data in enumerate(trainloader, 0):\n        inputs, labels = data\n        optimizer.zero_grad()\n        \n        outputs = net(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        \n        # print statistics\n        running_loss += loss.item()\n        if i % 200 == 199:\n            print('[%d, %5d] loss: %.3f' %\n                 (epoch + 1, i + 1, running_loss / 200)\n                 )\n            running_loss_list.append(running_loss)\n            running_loss = 0.0\nprint('Finished Training')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f10483590dfc7572fcdc2574b5463f4331a5e63f"},"cell_type":"code","source":"plt.plot(running_loss_list)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5463e9fbb9c7dfe2f88a0d0c6c94b376f832ebc0"},"cell_type":"markdown","source":"I't better, we will take this model as our final one."},{"metadata":{"_uuid":"2570c1d84788fed3b7d44bc101a0857d3f11b560"},"cell_type":"markdown","source":"## 5. Evaluate the model"},{"metadata":{"trusted":true,"_uuid":"293f197b1a05b62046215b209b92251c053de315"},"cell_type":"code","source":"correct = 0\ntotal = 0\nwith torch.no_grad():\n    for data in trainloader:\n        images, labels = data\n        outputs = net(images)\n        _, predicted = torch.max(outputs.data, 1)\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\nprint('Accuracy of the network on train images: ', correct/total)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a2ec4a40d3f2583dd8c77097be5d2754f5bc8a3b"},"cell_type":"markdown","source":"We achieve 98% accuracy on training data. It's not bad!"},{"metadata":{"_uuid":"733469b38e2cc0ea4e153606d3c8d41a6f640df7"},"cell_type":"markdown","source":"## 6. Prediction and Submition"},{"metadata":{"trusted":true,"_uuid":"b5f19ca8bcaf85edad8ad7ed076b4804e53858c8"},"cell_type":"code","source":"test_df = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a1c3eff58cec87418289390d071219a12d0da02"},"cell_type":"code","source":"test_df.values.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bbfe0bc9b874a61aa5a764689d668823be45052"},"cell_type":"code","source":"test_tensor = torch.tensor(test_df.values, dtype=torch.float) / 255\ntest_tensor = test_tensor.view(-1, 1, 28, 28)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf56f234315c05678db892ea658d8947b5c6a9df"},"cell_type":"code","source":"outputs = net(test_tensor)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"964c4cbca08a7e5bcd163e88d2027cba04c6197e"},"cell_type":"code","source":"_, predicted = torch.max(outputs, 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f675320f66227b6c4ae814d8df0b713df97b0f6f"},"cell_type":"code","source":"submit_df = pd.DataFrame({'ImageId': np.arange(1, 28001), 'Label': predicted.numpy()})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29b343e496d056590a107de379c52aaa87f4c25b"},"cell_type":"code","source":"submit_df.to_csv('cnn.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"12f2b94ec5d6657470f83a22eacb8985e1235c57"},"cell_type":"markdown","source":"### That's all, thank's for advice!"},{"metadata":{"trusted":true,"_uuid":"43a45ffdf9c2a3909c3b31e67b478cc1fd4e1295"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}