{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"font-family:Times New Roman;\"> <center> Pytorch Feedforward Network Tutorial</center> </h1>\n<p><center style=\"color:#159364; font-family:Times New Roman;font-size:30px;\">“The beautiful thing about learning is nobody can take it away from you.” — B.B. King</center></p>\n\n***","metadata":{}},{"cell_type":"markdown","source":"![Mnist](https://raw.githubusercontent.com/106AbdulBasit/Pytorch__Series/main/MNIST-%20FeedForward_Network/MnistExamples.png)","metadata":{}},{"cell_type":"markdown","source":"\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">In This Notebook you will learn </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Introduction to Neural Networks </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Some EDA </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Making the model in Pytorch </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Traing The model </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Losses and Accuracies </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Predictions </span>\n\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Submissions </span>\n\n","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:14px; font-family:Times New Roman; line-height: 1.7em;\">\n    📌 &nbsp;  If you find this notebook useful in anyway, please upvote it so that it can reach a bigger audience. You can share it with your fellow kagglers. 🙏🙏\n</div>","metadata":{}},{"cell_type":"markdown","source":"<h1  id=\"1\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\"> Introduction to Networks</h1>\n           \n   \n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> A detail descriptions of Neural Network with the reference of Brain precptrons will be added soon </span>\n","metadata":{}},{"cell_type":"markdown","source":"<h1  id=\"2\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\"> 2 Importing some Libraries </h1>\n\n\n\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np  \nfrom torch.utils.data import Dataset,DataLoader\nimport matplotlib.pyplot as plt \n%matplotlib inline\n\nimport torch\nimport torchvision\nimport numpy as np\nimport matplotlib\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torchvision.datasets import MNIST\nfrom torchvision.transforms import ToTensor\nfrom torchvision.utils import make_grid\nfrom torch.utils.data.dataloader import DataLoader\nfrom torch.utils.data import random_split\n%matplotlib inline\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:55:11.488776Z","iopub.execute_input":"2022-08-07T10:55:11.489286Z","iopub.status.idle":"2022-08-07T10:55:15.496410Z","shell.execute_reply.started":"2022-08-07T10:55:11.489193Z","shell.execute_reply":"2022-08-07T10:55:15.495367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n<h1  id=\"3\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\">3 Some EDA </h1>","metadata":{}},{"cell_type":"code","source":"trainset = pd.read_csv('../input/digit-recognizer/train.csv')\ntestset = pd.read_csv('../input/digit-recognizer/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:55:27.966276Z","iopub.execute_input":"2022-08-07T10:55:27.966683Z","iopub.status.idle":"2022-08-07T10:55:33.281482Z","shell.execute_reply.started":"2022-08-07T10:55:27.966649Z","shell.execute_reply":"2022-08-07T10:55:33.280538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:55:39.817036Z","iopub.execute_input":"2022-08-07T10:55:39.817444Z","iopub.status.idle":"2022-08-07T10:55:39.848230Z","shell.execute_reply.started":"2022-08-07T10:55:39.817412Z","shell.execute_reply":"2022-08-07T10:55:39.847293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainLabels = trainset.loc[:,'label']\ntrainSamples = trainset.loc[:,'pixel0':]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:55:42.972949Z","iopub.execute_input":"2022-08-07T10:55:42.973516Z","iopub.status.idle":"2022-08-07T10:55:42.982787Z","shell.execute_reply.started":"2022-08-07T10:55:42.973466Z","shell.execute_reply":"2022-08-07T10:55:42.981928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainSamples.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:55:45.223192Z","iopub.execute_input":"2022-08-07T10:55:45.223623Z","iopub.status.idle":"2022-08-07T10:55:47.573492Z","shell.execute_reply.started":"2022-08-07T10:55:45.223586Z","shell.execute_reply":"2022-08-07T10:55:47.572264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainLabels.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:55:51.502635Z","iopub.execute_input":"2022-08-07T10:55:51.503085Z","iopub.status.idle":"2022-08-07T10:55:51.516589Z","shell.execute_reply.started":"2022-08-07T10:55:51.503026Z","shell.execute_reply":"2022-08-07T10:55:51.515066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainLabels.head()\n#print(len(trainLabels))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:56:00.574753Z","iopub.execute_input":"2022-08-07T10:56:00.575207Z","iopub.status.idle":"2022-08-07T10:56:00.583480Z","shell.execute_reply.started":"2022-08-07T10:56:00.575168Z","shell.execute_reply":"2022-08-07T10:56:00.582207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Class_Id_Dist_Total = trainLabels.value_counts()\nClass_Id_Dist_Total.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:56:02.986561Z","iopub.execute_input":"2022-08-07T10:56:02.986967Z","iopub.status.idle":"2022-08-07T10:56:02.996311Z","shell.execute_reply.started":"2022-08-07T10:56:02.986933Z","shell.execute_reply":"2022-08-07T10:56:02.995109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nfig = go.Figure(go.Bar(\n            x= Class_Id_Dist_Total.values,\n            y=Class_Id_Dist_Total.index,\n            orientation='h'))\n\nfig.update_layout(title='Data Distribution in Bars',font_size=15,title_x=0.45)\n\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:56:07.572368Z","iopub.execute_input":"2022-08-07T10:56:07.572744Z","iopub.status.idle":"2022-08-07T10:56:07.678343Z","shell.execute_reply.started":"2022-08-07T10:56:07.572714Z","shell.execute_reply":"2022-08-07T10:56:07.677311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig=px.pie(Class_Id_Dist_Total.head(10),values= 'label', names=Class_Id_Dist_Total.index, hole=0.425)\nfig.update_layout(title='Data Distribution of Data',font_size=15,title_x=0.45,annotations=[dict(text='MNIST DATA SET',font_size=18, showarrow=False,height=800,width=700)])\nfig.update_traces(textfont_size=15,textinfo='percent')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:09.191522Z","iopub.execute_input":"2022-08-07T10:57:09.191971Z","iopub.status.idle":"2022-08-07T10:57:09.275258Z","shell.execute_reply.started":"2022-08-07T10:57:09.191934Z","shell.execute_reply":"2022-08-07T10:57:09.274013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainSamples /= 255","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:15.389645Z","iopub.execute_input":"2022-08-07T10:57:15.390763Z","iopub.status.idle":"2022-08-07T10:57:15.468540Z","shell.execute_reply.started":"2022-08-07T10:57:15.390693Z","shell.execute_reply":"2022-08-07T10:57:15.467492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(trainLabels))\nprint(len(trainSamples))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:18.663144Z","iopub.execute_input":"2022-08-07T10:57:18.663543Z","iopub.status.idle":"2022-08-07T10:57:18.669450Z","shell.execute_reply.started":"2022-08-07T10:57:18.663512Z","shell.execute_reply":"2022-08-07T10:57:18.668432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1  id=\"4\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\"> 4 Data Preprocessing </h1>","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import Dataset,DataLoader\n\nclass datasets(Dataset):\n    def __init__(self,images=trainSamples,labels = trainLabels):\n        self.images = images\n        self.labels = trainLabels\n        \n        self.len = len(images)\n        \n    def __len__(self):\n        return self.len \n    \n    def __getitem__(self,i):\n        \n        x = torch.tensor(np.array(trainSamples.loc[i,:]).reshape((28,28)),dtype=torch.float32).unsqueeze(0)\n        y = torch.tensor(np.array(self.labels[i]),dtype=torch.long)\n        return x,y","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:21.691651Z","iopub.execute_input":"2022-08-07T10:57:21.692069Z","iopub.status.idle":"2022-08-07T10:57:21.700484Z","shell.execute_reply.started":"2022-08-07T10:57:21.692019Z","shell.execute_reply":"2022-08-07T10:57:21.699217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = datasets()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:28.447003Z","iopub.execute_input":"2022-08-07T10:57:28.447420Z","iopub.status.idle":"2022-08-07T10:57:28.452971Z","shell.execute_reply.started":"2022-08-07T10:57:28.447387Z","shell.execute_reply":"2022-08-07T10:57:28.451528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">Let’s have a look at some of the images from the dataset. The images are converted to the Pytorch Tensors With the shape of 1x28x28 (Here 1 represents the dimensions means the color channel, and 28 x  28 represents the width and height). We will use plt.imshow to display the images. The permute method is used to reorder the dimensions of the image.  </span>","metadata":{}},{"cell_type":"code","source":"image, label = dataset[6]\nprint('image.shape:', image.shape)\nplt.imshow(image.permute(1, 2, 0), cmap='gray')\nprint('Label:', label)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:30.821403Z","iopub.execute_input":"2022-08-07T10:57:30.821854Z","iopub.status.idle":"2022-08-07T10:57:31.036321Z","shell.execute_reply.started":"2022-08-07T10:57:30.821816Z","shell.execute_reply":"2022-08-07T10:57:31.035356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dataset)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:34.119840Z","iopub.execute_input":"2022-08-07T10:57:34.120280Z","iopub.status.idle":"2022-08-07T10:57:34.126121Z","shell.execute_reply.started":"2022-08-07T10:57:34.120239Z","shell.execute_reply":"2022-08-07T10:57:34.125368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">We will use random_split helper function to set aside the images for the validation set. </span>","metadata":{}},{"cell_type":"code","source":"val_size = 10000\ntrain_size = len(dataset) - val_size\n\ntrain_ds, val_ds = random_split(dataset, [train_size, val_size])\nlen(train_ds), len(val_ds)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:36.687609Z","iopub.execute_input":"2022-08-07T10:57:36.688026Z","iopub.status.idle":"2022-08-07T10:57:36.697847Z","shell.execute_reply.started":"2022-08-07T10:57:36.687993Z","shell.execute_reply":"2022-08-07T10:57:36.696908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">Now we use Pytorch data loader function to create the data loader for the training and validation data set </span>","metadata":{}},{"cell_type":"code","source":"batch_size=128","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:40.488695Z","iopub.execute_input":"2022-08-07T10:57:40.489136Z","iopub.status.idle":"2022-08-07T10:57:40.494329Z","shell.execute_reply.started":"2022-08-07T10:57:40.489100Z","shell.execute_reply":"2022-08-07T10:57:40.493325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataLoader(train_ds, batch_size, shuffle=True, num_workers=4, pin_memory=True)\nval_loader = DataLoader(val_ds, batch_size*2, num_workers=4, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:45.258548Z","iopub.execute_input":"2022-08-07T10:57:45.259223Z","iopub.status.idle":"2022-08-07T10:57:45.265655Z","shell.execute_reply.started":"2022-08-07T10:57:45.259173Z","shell.execute_reply":"2022-08-07T10:57:45.264535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> **Num_ Workers** == how many subprocesses to use for data loading. 0 means that the data will be loaded in the main process. (default: 0) </span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">  **Pin_Memory** == If True, the data loader will copy Tensors into device/CUDA pinned memory before returning them. If your data elements are a custom type, or your collate_fn returns a batch that is a custom type, see the example below.</span>","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">Let visualize some of the images from the batch of the data in grid using  the make_grid function from torchvision</span>","metadata":{}},{"cell_type":"code","source":"for images, _ in train_loader:\n    print('images.shape:', images.shape)\n    plt.figure(figsize=(16,8))\n    plt.axis('off')\n    plt.imshow(make_grid(images, nrow=16).permute((1, 2, 0)))\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:48.502595Z","iopub.execute_input":"2022-08-07T10:57:48.503249Z","iopub.status.idle":"2022-08-07T10:57:49.076390Z","shell.execute_reply.started":"2022-08-07T10:57:48.503210Z","shell.execute_reply":"2022-08-07T10:57:49.075122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">We will create the batch of input tensors. In order to do that, We‘ll flatten the 1x28x28 images into vectors of size 784, so they can be passed into an nn.Linear object.\n\n </span>","metadata":{}},{"cell_type":"code","source":"for images, labels in train_loader:\n    print('images.shape:', images.shape)\n    inputs = images.reshape(-1, 784)\n    print('inputs.shape:', inputs.shape)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:52.978872Z","iopub.execute_input":"2022-08-07T10:57:52.979765Z","iopub.status.idle":"2022-08-07T10:57:53.208963Z","shell.execute_reply.started":"2022-08-07T10:57:52.979713Z","shell.execute_reply":"2022-08-07T10:57:53.207750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">Next, let's create a `nn.Linear` object, which will serve as our _hidden_ layer. We'll set the size of the output from the hidden layer to 32. This number can be increased or decreased to change the _learning capacity_ of the model. </span>","metadata":{}},{"cell_type":"code","source":"input_size = inputs.shape[-1]\nhidden_size = 32","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:56.965780Z","iopub.execute_input":"2022-08-07T10:57:56.966906Z","iopub.status.idle":"2022-08-07T10:57:56.971946Z","shell.execute_reply.started":"2022-08-07T10:57:56.966863Z","shell.execute_reply":"2022-08-07T10:57:56.970929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">**Question** : Why we are using [-1] </span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Answer : -1 means the last index in the list. In our example, it points to 784 </span>","metadata":{}},{"cell_type":"markdown","source":"<h1  id=\"5\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\"> 4 Linear Layers and activation Functions </h1>","metadata":{}},{"cell_type":"code","source":"layer1 = nn.Linear(input_size, hidden_size)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:57:59.966034Z","iopub.execute_input":"2022-08-07T10:57:59.966447Z","iopub.status.idle":"2022-08-07T10:57:59.972785Z","shell.execute_reply.started":"2022-08-07T10:57:59.966416Z","shell.execute_reply":"2022-08-07T10:57:59.971617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:04.192116Z","iopub.execute_input":"2022-08-07T10:58:04.192489Z","iopub.status.idle":"2022-08-07T10:58:04.200040Z","shell.execute_reply.started":"2022-08-07T10:58:04.192457Z","shell.execute_reply":"2022-08-07T10:58:04.198816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">We can now compute intermediate outputs for the batch of images by passing `inputs` through `layer1`. </span>\n\n","metadata":{}},{"cell_type":"code","source":"layer1_outputs = layer1(inputs)\nprint('layer1_outputs.shape:', layer1_outputs.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:12.159131Z","iopub.execute_input":"2022-08-07T10:58:12.160127Z","iopub.status.idle":"2022-08-07T10:58:12.183446Z","shell.execute_reply.started":"2022-08-07T10:58:12.160085Z","shell.execute_reply":"2022-08-07T10:58:12.182218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">The image vectors of size `784` are transformed into intermediate output vectors of length `32` by performing a matrix multiplication of `inputs` matrix with the transposed weights matrix of `layer1` and adding the bias. We can verify this using `torch.allclose` </span>","metadata":{}},{"cell_type":"code","source":"layer1_outputs_direct = inputs @ layer1.weight.t() + layer1.bias\nlayer1_outputs_direct.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:15.802240Z","iopub.execute_input":"2022-08-07T10:58:15.802612Z","iopub.status.idle":"2022-08-07T10:58:15.810174Z","shell.execute_reply.started":"2022-08-07T10:58:15.802580Z","shell.execute_reply":"2022-08-07T10:58:15.809315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.allclose(layer1_outputs, layer1_outputs_direct, 1e-3)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:22.209614Z","iopub.execute_input":"2022-08-07T10:58:22.210006Z","iopub.status.idle":"2022-08-07T10:58:22.220206Z","shell.execute_reply.started":"2022-08-07T10:58:22.209974Z","shell.execute_reply":"2022-08-07T10:58:22.218976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">Thus, `layer1_outputs` and `inputs` have a linear relationship, i.e., each element of `layer_outputs` is a weighted sum of elements from `inputs`. Thus, even as we train the model and modify the weights, `layer1` can only capture linear relationships between `inputs` and `outputs`.\n\n<img src=\"https://i.imgur.com/inXsLuq.png\" width=\"360\"> </span>\n","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> Next, we'll use the Rectified Linear Unit (ReLU) function as the activation function for the outputs. It has the formula `relu(x) = max(0,x)` i.e. it simply replaces negative values in a given tensor with the value 0. ReLU is a non-linear function, as seen here visually: </span>\n\n\n<img src=\"https://i.imgur.com/yijV4xF.png\" width=\"420\">\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> We can use the `F.relu` method to apply ReLU to the elements of a tensor.</span>\n\n","metadata":{}},{"cell_type":"code","source":"F.relu(torch.tensor([[1, -1, 0], \n                     [-0.1, .2, 3]]))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:26.686987Z","iopub.execute_input":"2022-08-07T10:58:26.688170Z","iopub.status.idle":"2022-08-07T10:58:26.698264Z","shell.execute_reply.started":"2022-08-07T10:58:26.688131Z","shell.execute_reply":"2022-08-07T10:58:26.697179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> lets apply relu activation functoion.You can see it change the value from negative to zero. it does not change the entire value into positive . it just convert that negative value to minimum positive value which is 0 </span>","metadata":{}},{"cell_type":"code","source":"relu_outputs = F.relu(layer1_outputs)\nprint('min(layer1_outputs):', torch.min(layer1_outputs).item())\nprint('min(relu_outputs):', torch.min(relu_outputs).item())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:29.987568Z","iopub.execute_input":"2022-08-07T10:58:29.988003Z","iopub.status.idle":"2022-08-07T10:58:29.994374Z","shell.execute_reply.started":"2022-08-07T10:58:29.987969Z","shell.execute_reply":"2022-08-07T10:58:29.993466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">Next, let's create an output layer to convert vectors of length hidden_size in relu_outputs into vectors of length 10, which is the desired output of our model (since there are 10 target labels). </span>","metadata":{}},{"cell_type":"code","source":"output_size = 10\nlayer2 = nn.Linear(hidden_size, output_size)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:37.236496Z","iopub.execute_input":"2022-08-07T10:58:37.236939Z","iopub.status.idle":"2022-08-07T10:58:37.242515Z","shell.execute_reply.started":"2022-08-07T10:58:37.236901Z","shell.execute_reply":"2022-08-07T10:58:37.241576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">We applied the linear function. In a moment you will understand. </span>","metadata":{}},{"cell_type":"code","source":"layer2_outputs = layer2(relu_outputs)\nprint(layer2_outputs.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:40.465302Z","iopub.execute_input":"2022-08-07T10:58:40.465696Z","iopub.status.idle":"2022-08-07T10:58:40.474133Z","shell.execute_reply.started":"2022-08-07T10:58:40.465666Z","shell.execute_reply":"2022-08-07T10:58:40.473026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:43.702615Z","iopub.execute_input":"2022-08-07T10:58:43.703016Z","iopub.status.idle":"2022-08-07T10:58:43.710491Z","shell.execute_reply.started":"2022-08-07T10:58:43.702984Z","shell.execute_reply":"2022-08-07T10:58:43.709131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">As expected, `layer2_outputs` contains a batch of vectors of size 10. We can now use this output to compute the loss using `F.cross_entropy` and adjust the weights of `layer1` and `layer2` using gradient descent.</span>","metadata":{}},{"cell_type":"code","source":"F.cross_entropy(layer2_outputs, labels)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:47.034405Z","iopub.execute_input":"2022-08-07T10:58:47.034835Z","iopub.status.idle":"2022-08-07T10:58:47.046762Z","shell.execute_reply.started":"2022-08-07T10:58:47.034800Z","shell.execute_reply":"2022-08-07T10:58:47.045535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> Thus, our model transforms `inputs` into `layer2_outputs` by applying a linear transformation (using `layer1`), followed by a non-linear activation (using `F.relu`), followed by another linear transformation (using `layer2`). Let's verify this by re-computing the output using basic matrix operations. </span>","metadata":{}},{"cell_type":"code","source":"torch.allclose(outputs, layer2_outputs, 1e-3)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:14:36.811884Z","iopub.execute_input":"2022-08-03T11:14:36.812346Z","iopub.status.idle":"2022-08-03T11:14:36.821965Z","shell.execute_reply.started":"2022-08-03T11:14:36.812296Z","shell.execute_reply":"2022-08-03T11:14:36.820526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Same as layer2(layer1(inputs))\noutputs2 = (inputs @ layer1.weight.t() + layer1.bias) @ layer2.weight.t() + layer2.bias","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:15:03.151176Z","iopub.execute_input":"2022-08-03T11:15:03.151961Z","iopub.status.idle":"2022-08-03T11:15:03.163659Z","shell.execute_reply.started":"2022-08-03T11:15:03.151921Z","shell.execute_reply":"2022-08-03T11:15:03.162530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1  id=\"6\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\">6 Model Making </h1>","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\">We are now ready to define our model. As discussed above, we'll create a neural network with one hidden layer. Here's what that means:</span>\n\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">* Instead of using a single `nn.Linear` object to transform a batch of inputs (pixel intensities) into outputs (class probabilities), we'll use two `nn.Linear` objects. Each of these is called a _layer_ in the network.  </span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">* The first layer (also known as the hidden layer) will transform the input matrix of shape `batch_size x 784` into an intermediate output matrix of shape `batch_size x hidden_size`. The parameter `hidden_size` can be configured manually (e.g., 32 or 64).</span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> * We'll then apply a non-linear *activation function* to the intermediate outputs. The activation function transforms individual elements of the matrix. </span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> * The result of the activation function, which is also of size `batch_size x hidden_size`, is passed into the second layer (also known as the output layer).  The second layer transforms it into a matrix of size `batch_size x 10`. We can use this output to compute the loss and adjust weights using gradient descent.</span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> As discussed above, our model will contain one hidden layer. Here's what it looks like visually:\n </span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">  Let's define the model by extending the `nn.Module` class from PyTorch </span>\n\n\n\n<img src=\"https://i.imgur.com/eN7FrpF.png\" width=\"480\">\n\n\n.","metadata":{}},{"cell_type":"code","source":"class MnistModel(nn.Module):\n    \"\"\"Feedfoward neural network with 1 hidden layer\"\"\"\n    def __init__(self, in_size, hidden_size, out_size):\n        super().__init__()\n        # hidden layer\n        self.linear1 = nn.Linear(in_size, hidden_size)\n        # output layer\n        self.linear2 = nn.Linear(hidden_size, out_size)\n        \n    def forward(self, xb):\n        # Flatten the image tensors\n        xb = xb.view(xb.size(0), -1)\n        # Get intermediate outputs using hidden layer\n        out = self.linear1(xb)\n        # Apply activation function\n        out = F.relu(out)\n        # Get predictions using output layer\n        out = self.linear2(out)\n        return out\n    \n    def training_step(self, batch):\n        images, labels = batch \n        out = self(images)                  # Generate predictions\n        loss = F.cross_entropy(out, labels) # Calculate loss\n        return loss\n    \n    def validation_step(self, batch):\n        images, labels = batch \n        out = self(images)                    # Generate predictions\n        loss = F.cross_entropy(out, labels)   # Calculate loss\n        acc = accuracy(out, labels)           # Calculate accuracy\n        return {'val_loss': loss, 'val_acc': acc}\n        \n    def validation_epoch_end(self, outputs):\n        batch_losses = [x['val_loss'] for x in outputs]\n        epoch_loss = torch.stack(batch_losses).mean()   # Combine losses\n        batch_accs = [x['val_acc'] for x in outputs]\n        epoch_acc = torch.stack(batch_accs).mean()      # Combine accuracies\n        return {'val_loss': epoch_loss.item(), 'val_acc': epoch_acc.item()}\n    \n    def epoch_end(self, epoch, result):\n        print(\"Epoch [{}], val_loss: {:.4f}, val_acc: {:.4f}\".format(epoch, result['val_loss'], result['val_acc']))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:58:58.521565Z","iopub.execute_input":"2022-08-07T10:58:58.522014Z","iopub.status.idle":"2022-08-07T10:58:58.535144Z","shell.execute_reply.started":"2022-08-07T10:58:58.521977Z","shell.execute_reply":"2022-08-07T10:58:58.534109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> We also need to define an `accuracy` function which calculates the accuracy of the model's prediction on an batch of inputs. It's used in `validation_step` above. </span>","metadata":{}},{"cell_type":"code","source":"def accuracy(outputs, labels):\n    _, preds = torch.max(outputs, dim=1)\n    return torch.tensor(torch.sum(preds == labels).item() / len(preds))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:06.353508Z","iopub.execute_input":"2022-08-07T10:59:06.353997Z","iopub.status.idle":"2022-08-07T10:59:06.359816Z","shell.execute_reply.started":"2022-08-07T10:59:06.353960Z","shell.execute_reply":"2022-08-07T10:59:06.359010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> We'll create a model that contains a hidden layer with 32 activations. </span>","metadata":{}},{"cell_type":"code","source":"input_size = 784\nhidden_size = 32 # you can change this\nnum_classes = 10","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:10.046511Z","iopub.execute_input":"2022-08-07T10:59:10.046941Z","iopub.status.idle":"2022-08-07T10:59:10.052789Z","shell.execute_reply.started":"2022-08-07T10:59:10.046904Z","shell.execute_reply":"2022-08-07T10:59:10.051226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = MnistModel(input_size, hidden_size=32, out_size=num_classes)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:10:58.355515Z","iopub.execute_input":"2022-08-06T11:10:58.355951Z","iopub.status.idle":"2022-08-06T11:10:58.363270Z","shell.execute_reply.started":"2022-08-06T11:10:58.355914Z","shell.execute_reply":"2022-08-06T11:10:58.362186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<h1  id=\"7\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\"> 7 Using GPU </h1>","metadata":{}},{"cell_type":"code","source":"torch.cuda.is_available()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:13.111169Z","iopub.execute_input":"2022-08-07T10:59:13.111562Z","iopub.status.idle":"2022-08-07T10:59:13.118595Z","shell.execute_reply.started":"2022-08-07T10:59:13.111531Z","shell.execute_reply":"2022-08-07T10:59:13.117797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-family:Times New Roman; font-size:18px;\"> Let's define a helper function to ensure that our code uses the GPU if available and defaults to using the CPU if it isn't.  </span>\n","metadata":{}},{"cell_type":"code","source":"def get_default_device():\n    \"\"\"Pick GPU if available, else CPU\"\"\"\n    if torch.cuda.is_available():\n        return torch.device('cuda')\n    else:\n        return torch.device('cpu')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:15.890982Z","iopub.execute_input":"2022-08-07T10:59:15.891758Z","iopub.status.idle":"2022-08-07T10:59:15.898266Z","shell.execute_reply.started":"2022-08-07T10:59:15.891713Z","shell.execute_reply":"2022-08-07T10:59:15.897063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = get_default_device()\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:18.659190Z","iopub.execute_input":"2022-08-07T10:59:18.659579Z","iopub.status.idle":"2022-08-07T10:59:18.666498Z","shell.execute_reply.started":"2022-08-07T10:59:18.659547Z","shell.execute_reply":"2022-08-07T10:59:18.665385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-family:Times New Roman; font-size:18px;\"> Next, let's define a function that can move data and model to a chosen device.</span>\n","metadata":{}},{"cell_type":"code","source":"def to_device(data, device):\n    \"\"\"Move tensor(s) to chosen device\"\"\"\n    if isinstance(data, (list,tuple)):\n        return [to_device(x, device) for x in data]\n    return data.to(device, non_blocking=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:22.549402Z","iopub.execute_input":"2022-08-07T10:59:22.549827Z","iopub.status.idle":"2022-08-07T10:59:22.555928Z","shell.execute_reply.started":"2022-08-07T10:59:22.549793Z","shell.execute_reply":"2022-08-07T10:59:22.554731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for images, labels in train_loader:\n    print(images.shape)\n    images = to_device(images, device)\n    print(images.device)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:26.533537Z","iopub.execute_input":"2022-08-07T10:59:26.533969Z","iopub.status.idle":"2022-08-07T10:59:26.767176Z","shell.execute_reply.started":"2022-08-07T10:59:26.533937Z","shell.execute_reply":"2022-08-07T10:59:26.766035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-family:Times New Roman; font-size:18px;\"> Finally, we define a `DeviceDataLoader` class to wrap our existing data loaders and move batches of data to the selected device. Interestingly, we don't need to extend an existing class to create a PyTorch datal oader. All we need is an `__iter__` method to retrieve batches of data and an `__len__` method to get the number of batches.</span>\n","metadata":{}},{"cell_type":"code","source":"class DeviceDataLoader():\n    \"\"\"Wrap a dataloader to move data to a device\"\"\"\n    def __init__(self, dl, device):\n        self.dl = dl\n        self.device = device\n        \n    def __iter__(self):\n        \"\"\"Yield a batch of data after moving it to device\"\"\"\n        for b in self.dl: \n            yield to_device(b, self.device)\n\n    def __len__(self):\n        \"\"\"Number of batches\"\"\"\n        return len(self.dl)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:33.086131Z","iopub.execute_input":"2022-08-07T10:59:33.086593Z","iopub.status.idle":"2022-08-07T10:59:33.094795Z","shell.execute_reply.started":"2022-08-07T10:59:33.086542Z","shell.execute_reply":"2022-08-07T10:59:33.093744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-family:Times New Roman; font-size:18px;\"> The `yield` keyword in Python is used to create a generator function that can be used within a `for` loop, as illustrated below.</span>\n","metadata":{}},{"cell_type":"code","source":"def some_numbers():\n    yield 10\n    yield 20\n    yield 30\n\nfor value in some_numbers():\n    print(value)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:35.944402Z","iopub.execute_input":"2022-08-07T10:59:35.944854Z","iopub.status.idle":"2022-08-07T10:59:35.951012Z","shell.execute_reply.started":"2022-08-07T10:59:35.944822Z","shell.execute_reply":"2022-08-07T10:59:35.949855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DeviceDataLoader(train_loader, device)\nval_loader = DeviceDataLoader(val_loader, device)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:39.061805Z","iopub.execute_input":"2022-08-07T10:59:39.062620Z","iopub.status.idle":"2022-08-07T10:59:39.068803Z","shell.execute_reply.started":"2022-08-07T10:59:39.062581Z","shell.execute_reply":"2022-08-07T10:59:39.067490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-family:Times New Roman; font-size:18px;\"> Tensors moved to the GPU have a `device` property which includes that word `cuda`. Let's verify this by looking at a batch of data from `valid_dl`. </span>\n","metadata":{}},{"cell_type":"code","source":"for xb, yb in val_loader:\n    print('xb.device:', xb.device)\n    print('yb:', yb)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-07T10:59:44.053158Z","iopub.execute_input":"2022-08-07T10:59:44.053576Z","iopub.status.idle":"2022-08-07T10:59:44.374517Z","shell.execute_reply.started":"2022-08-07T10:59:44.053540Z","shell.execute_reply":"2022-08-07T10:59:44.373119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1  id=\"8\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\">  8 Training The Model </h1>","metadata":{}},{"cell_type":"markdown","source":"<span style=\"font-family:Times New Roman; font-size:18px;\"> \n\nWe'll define two functions: `fit` and `evaluate` to train the model using gradient descent and evaluate its performance on the validation set.</span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">  Consider the following Image. It is the illustration that how learning  works.</span>\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">   On your left side you see the graph in whihc cost function is finding minimum value of loss . The more it value goes deep down the more accurate the line fits on the right side.</span>","metadata":{}},{"cell_type":"markdown","source":"![Gradient](https://raw.githubusercontent.com/106AbdulBasit/Pytorch__Series/main/MNIST-%20FeedForward_Network/Gradient.gif)","metadata":{}},{"cell_type":"code","source":"def evaluate(model, val_loader):\n    \"\"\"Evaluate the model's performance on the validation set\"\"\"\n    outputs = [model.validation_step(batch) for batch in val_loader]\n    return model.validation_epoch_end(outputs)\n\ndef fit(epochs, lr, model, train_loader, val_loader, opt_func=torch.optim.SGD):\n    \"\"\"Train the model using gradient descent\"\"\"\n    history = []\n    optimizer = opt_func(model.parameters(), lr)\n    for epoch in range(epochs):\n        # Training Phase \n        for batch in train_loader:\n            loss = model.training_step(batch)\n            loss.backward()\n            optimizer.step()\n            optimizer.zero_grad()\n        # Validation phase\n        result = evaluate(model, val_loader)\n        model.epoch_end(epoch, result)\n        history.append(result)\n    return history","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:02:00.122505Z","iopub.execute_input":"2022-08-07T11:02:00.123366Z","iopub.status.idle":"2022-08-07T11:02:00.131038Z","shell.execute_reply.started":"2022-08-07T11:02:00.123320Z","shell.execute_reply":"2022-08-07T11:02:00.130268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model (on GPU)\nmodel = MnistModel(input_size, hidden_size=hidden_size, out_size=num_classes)\nto_device(model, device)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:02:03.179880Z","iopub.execute_input":"2022-08-07T11:02:03.180614Z","iopub.status.idle":"2022-08-07T11:02:03.188617Z","shell.execute_reply.started":"2022-08-07T11:02:03.180576Z","shell.execute_reply":"2022-08-07T11:02:03.187654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = [evaluate(model, val_loader)]\nhistory","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:02:05.459198Z","iopub.execute_input":"2022-08-07T11:02:05.460393Z","iopub.status.idle":"2022-08-07T11:02:06.507838Z","shell.execute_reply.started":"2022-08-07T11:02:05.460328Z","shell.execute_reply":"2022-08-07T11:02:06.506544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history += fit(5, 0.5, model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:02:11.053625Z","iopub.execute_input":"2022-08-07T11:02:11.054083Z","iopub.status.idle":"2022-08-07T11:02:34.232684Z","shell.execute_reply.started":"2022-08-07T11:02:11.054031Z","shell.execute_reply":"2022-08-07T11:02:34.231674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history += fit(5, 0.1, model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:02:44.982898Z","iopub.execute_input":"2022-08-07T11:02:44.983384Z","iopub.status.idle":"2022-08-07T11:03:08.192844Z","shell.execute_reply.started":"2022-08-07T11:02:44.983342Z","shell.execute_reply":"2022-08-07T11:03:08.191507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Now we plot the losses and the accuracies of the model. </span>","metadata":{}},{"cell_type":"code","source":"losses = [x['val_loss'] for x in history]\nplt.plot(losses, '-x')\nplt.xlabel('epoch')\nplt.ylabel('loss')\nplt.title('Loss vs. No. of epochs');","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:12.026500Z","iopub.execute_input":"2022-08-07T11:03:12.027123Z","iopub.status.idle":"2022-08-07T11:03:12.224552Z","shell.execute_reply.started":"2022-08-07T11:03:12.027068Z","shell.execute_reply":"2022-08-07T11:03:12.223447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracies = [x['val_acc'] for x in history]\nplt.plot(accuracies, '-x')\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\nplt.title('Accuracy vs. No. of epochs');","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:15.043306Z","iopub.execute_input":"2022-08-07T11:03:15.043714Z","iopub.status.idle":"2022-08-07T11:03:15.240781Z","shell.execute_reply.started":"2022-08-07T11:03:15.043671Z","shell.execute_reply":"2022-08-07T11:03:15.239522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1  id=\"9\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\">  8 Predictions</h1>","metadata":{}},{"cell_type":"code","source":"def prediction(model, loader, device=\"cpu\"):\n    model.to(device)\n    model.eval()\n    preds_all = torch.LongTensor()\n    \n    with torch.no_grad():\n        for images in loader:\n            images = images.to(device)\n            \n            output = model.forward(images)            \n            probs = torch.exp(output)\n            pred = probs.to('cpu').max(dim=1)[1]\n            preds_all = torch.cat((preds_all, pred), dim=0)\n    return preds_all","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:18.759147Z","iopub.execute_input":"2022-08-07T11:03:18.759528Z","iopub.status.idle":"2022-08-07T11:03:18.767375Z","shell.execute_reply.started":"2022-08-07T11:03:18.759498Z","shell.execute_reply":"2022-08-07T11:03:18.766133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = testset.values.reshape(-1,1,28,28)\nprint(\"Shape of test set: {}\".format(test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:21.451607Z","iopub.execute_input":"2022-08-07T11:03:21.452002Z","iopub.status.idle":"2022-08-07T11:03:21.457751Z","shell.execute_reply.started":"2022-08-07T11:03:21.451970Z","shell.execute_reply":"2022-08-07T11:03:21.456534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tensor = torch.tensor(test)/255.0","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:25.571971Z","iopub.execute_input":"2022-08-07T11:03:25.572420Z","iopub.status.idle":"2022-08-07T11:03:25.787519Z","shell.execute_reply.started":"2022-08-07T11:03:25.572385Z","shell.execute_reply":"2022-08-07T11:03:25.786252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = DataLoader(test_tensor, batch_size=16)\ny_pred = prediction(model, test_loader)\nprint(\"Prediction completed...\")\n\n# Creating a dataframe for results\ntest = pd.read_csv(\"../input/digit-recognizer/test.csv\")\nresult = pd.DataFrame({'ImageId': test.index, 'Label': y_pred})\nresult[\"ImageId\"] += 1\nresult.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:27.606754Z","iopub.execute_input":"2022-08-07T11:03:27.607640Z","iopub.status.idle":"2022-08-07T11:03:29.651345Z","shell.execute_reply.started":"2022-08-07T11:03:27.607569Z","shell.execute_reply":"2022-08-07T11:03:29.650221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot images from test data and see predictions\nindices = [41, 314, 2022, 22111]\nf,ax = plt.subplots(1, len(indices))\nfor i in range(len(indices)):\n    title = \"Label: {}\".format(int(y_pred[indices[i]]))\n    ax[i].imshow( test_tensor[indices[i],0] )\n    ax[i].set_title(title)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T11:03:32.880902Z","iopub.execute_input":"2022-08-07T11:03:32.881925Z","iopub.status.idle":"2022-08-07T11:03:33.265994Z","shell.execute_reply.started":"2022-08-07T11:03:32.881886Z","shell.execute_reply":"2022-08-07T11:03:33.265093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.to_csv('submission.csv', index=False)\nprint(\"Resuls are saved to submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T10:51:19.591157Z","iopub.execute_input":"2022-08-06T10:51:19.592356Z","iopub.status.idle":"2022-08-06T10:51:19.631230Z","shell.execute_reply.started":"2022-08-06T10:51:19.592308Z","shell.execute_reply":"2022-08-06T10:51:19.629944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"font-size:14px; font-family:Times New Roman; line-height: 1.7em;\">\n    📌 &nbsp;  If you find this notebook useful in anyway, please upvote it so that it can reach a bigger audience. You can share it with your fellow kagglers. 🙏🙏\n</div>","metadata":{}},{"cell_type":"markdown","source":"<h1  id=\"10\" style=\"color:black;\n           display:fill;\n           border-radius:5px;\n           background-color:#FFFFFF;\n           font-size:350%;\n           font-family:Times New Roman;\n           letter-spacing:0.5px\"> Whats Next??</h1>","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [Linear Regression _ Pytorch For Beginner's](https://www.kaggle.com/code/abdulbasitniazi/linear-regression-pytorch-for-beginner-s) </span> \n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [ENetB7 Explained 98% ✅ Fine Tuning ✅ EDA✅](https://www.kaggle.com/code/abdulbasitniazi/enetb7-explained-98-fine-tuning-eda) </span> \n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [ResNet50_✅EDA✅Transfer Learning](https://www.kaggle.com/code/abdulbasitniazi/resnet50-eda-transfer-learning)</span> \n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [ResNet50FromScratch📊 EDA 🐴🐆🐄🐏🐕](https://www.kaggle.com/code/abdulbasitniazi/resnet50fromscratch-eda)</span> \n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [🐶Vs😻✅ How CNN Works? ✅EDA✅](https://www.kaggle.com/code/abdulbasitniazi/vs-how-cnn-works-eda)</span> \n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [Medium Link](https://medium.com/@AB_Niazi)</span> \n\n<span style=\"font-family:Times New Roman; font-size:18px;\"> [For Video Explanation  Youtube  Link](https://www.youtube.com/channel/UCSAw-QDHdXjrAMpg7-TELVA)</span> \n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"\n\n\n<span style=\"font-family:Times New Roman; font-size:18px;\">Refernces </span>\n\n[Akaash Tutorial on Jovian](https://jovian.ai/aakashns/04-feedforward-nn)\n\n","metadata":{}}]}