{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torch.utils.data as data\n\nimport torchvision.transforms as transforms\nimport torchvision.datasets as datasets\nfrom torchvision.utils import make_grid\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport copy,os,PIL\nimport random\nimport time","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T05:24:22.523549Z","iopub.execute_input":"2022-07-23T05:24:22.523909Z","iopub.status.idle":"2022-07-23T05:24:22.530937Z","shell.execute_reply.started":"2022-07-23T05:24:22.523880Z","shell.execute_reply":"2022-07-23T05:24:22.529846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed everything here. Since there is no actual randomization for computer we need to set a random state in order to produce identical results \n# no matter how many times we run the code.\nSEED = 2022\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\ntorch.cuda.manual_seed(SEED)\ntorch.backends.cudnn.deterministic = True","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:24:24.103868Z","iopub.execute_input":"2022-07-23T05:24:24.104262Z","iopub.status.idle":"2022-07-23T05:24:24.111760Z","shell.execute_reply.started":"2022-07-23T05:24:24.104226Z","shell.execute_reply":"2022-07-23T05:24:24.110634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating a custom_dataset class for loading.\nclass Custom_Dataset(torch.utils.data.Dataset):\n\n    def __init__(self, image_dir, csv_file, transform=None):\n        self.image_dir = image_dir\n        self.data = pd.read_csv(csv_file, header=0)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, index):\n        image_name = os.path.join(self.image_dir, self.data.loc[index, 'id'])  \n        image = PIL.Image.open(image_name)\n        label = self.data.loc[index, 'label']\n        if self.transform:\n            image = self.transform(image)\n        return (image, label)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:25:20.014467Z","iopub.execute_input":"2022-07-23T05:25:20.014853Z","iopub.status.idle":"2022-07-23T05:25:20.023097Z","shell.execute_reply.started":"2022-07-23T05:25:20.014823Z","shell.execute_reply":"2022-07-23T05:25:20.021612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = '../input/msoc/ChineseMNIST/train/train/'\ncsv_file = '../input/msoc/ChineseMNIST/train.csv'\n# initialize a transformation object\ntransform = transforms.Compose([transforms.Resize((28,28)),\n                               transforms.ToTensor(),\n                               transforms.Normalize((0.5,), (0.5,))\n                               ])\n# load the custom dataset\ndataset = Custom_Dataset(image_dir, csv_file, transform=transform_img)\n# split the data into train and val set at the ratio of 0.8.\ntrain_percent=0.8\ntrain_size = int(train_percent * len(dataset))\nval_size = len(dataset) - train_size\ntrain_data, val_data = torch.utils.data.random_split(dataset, [train_size, val_size])\n# set the batch and create dataLoaders.\nbatch_size=64\ntrain_loader = torch.utils.data.DataLoader(train_data, batch_size = batch_size, num_workers = 0)\nval_loader = torch.utils.data.DataLoader(val_data, batch_size = batch_size, num_workers = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:25:39.388659Z","iopub.execute_input":"2022-07-23T05:25:39.389033Z","iopub.status.idle":"2022-07-23T05:25:39.408958Z","shell.execute_reply.started":"2022-07-23T05:25:39.389003Z","shell.execute_reply":"2022-07-23T05:25:39.407511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_batch(dl):\n  for images, label in dl:\n    print('The shape of the image is:',images.shape)\n    fig, ax = plt.subplots(figsize=(12, 10))\n    ax.set_xticks([]); ax.set_yticks([])\n    ax.imshow(make_grid(images, nrow=8).permute(1, 2, 0))\n    break","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:26:21.640463Z","iopub.execute_input":"2022-07-23T05:26:21.640892Z","iopub.status.idle":"2022-07-23T05:26:21.647219Z","shell.execute_reply.started":"2022-07-23T05:26:21.640855Z","shell.execute_reply":"2022-07-23T05:26:21.646071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_batch(train_loader)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:26:23.333585Z","iopub.execute_input":"2022-07-23T05:26:23.333987Z","iopub.status.idle":"2022-07-23T05:26:23.625745Z","shell.execute_reply.started":"2022-07-23T05:26:23.333954Z","shell.execute_reply":"2022-07-23T05:26:23.624925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# simple baseline model which contains FC three layers\nclass MLP(nn.Module):\n    def __init__(self, input_dim, output_dim):\n        super().__init__()\n                \n        self.input_fc = nn.Linear(input_dim, 250)\n        self.hidden_fc = nn.Linear(250, 100)\n        self.output_fc = nn.Linear(100, output_dim)\n        \n    def forward(self, x):\n        \n        #x = [batch size, height, width]\n        \n        batch_size = x.shape[0]\n\n        x = x.view(batch_size, -1)\n        \n        #x = [batch size, height * width]\n        \n        h_1 = F.relu(self.input_fc(x))\n        \n        #h_1 = [batch size, 250]\n\n        h_2 = F.relu(self.hidden_fc(h_1))\n\n        #h_2 = [batch size, 100]\n\n        y_pred = self.output_fc(h_2)\n        \n        #y_pred = [batch size, output dim]\n        \n        return y_pred, h_2\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:28:03.808836Z","iopub.execute_input":"2022-07-23T05:28:03.809454Z","iopub.status.idle":"2022-07-23T05:28:03.816164Z","shell.execute_reply.started":"2022-07-23T05:28:03.809415Z","shell.execute_reply":"2022-07-23T05:28:03.815299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LeNet(nn.Module):\n    def __init__(self, output_dim):\n        super().__init__()\n        \n        ##### TODO #####\n        \n        # Conv layer1\n\n        # Conv layer2\n\n        # Fully connected layer1\n\n        # We use dropout layer between these both FCL as they have the highest number of parameters b/t them\n\n        # Fully connected layer2\n        \n        ##### END #####\n\n    def forward(self, x):\n        \n        ##### TODO #####\n        \n        # Apply ReLu to the feature maps produced after Conv 1 layer\n\n        # Pooling layer after Conv 1 layer\n\n        # Apply ReLu to the feature maps produced after Conv 2 layer\n\n        # Pooling layer after Conv 2 layer\n\n        # Flattening the output of CNN to feed it into Fully connected layer\n\n        # Fully connected layer 1 with Relu\n\n        # We use dropout layer between these both FCL as they have the highest number of parameters b/t them\n\n        # Fully connected layer 2 with no activation funct as we need raw output from CrossEntropyLoss\n\n        ##### END #####\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:11.618017Z","iopub.execute_input":"2022-07-23T05:29:11.618429Z","iopub.status.idle":"2022-07-23T05:29:11.624640Z","shell.execute_reply.started":"2022-07-23T05:29:11.618391Z","shell.execute_reply":"2022-07-23T05:29:11.623341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define the input dimension\nINPUT_DIM = 28 * 28\n# define the output dimension\nOUTPUT_DIM = 15\n# initialize our model\nmodel = MLP(INPUT_DIM, OUTPUT_DIM)\n# model = LeNet(OUTPUT_DIM)\n# set a optimizer for our model\noptimizer = optim.Adam(model.parameters())\n# define a loss function which is Cross Entropy Loss for this case\ncriterion = nn.CrossEntropyLoss()\n# find a device to deploy computation\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n# deploy model to device\nmodel = model.to(device)\ncriterion = criterion.to(device)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:33.467226Z","iopub.execute_input":"2022-07-23T05:29:33.467642Z","iopub.status.idle":"2022-07-23T05:29:33.478804Z","shell.execute_reply.started":"2022-07-23T05:29:33.467609Z","shell.execute_reply":"2022-07-23T05:29:33.477600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print the model architecture\nprint(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:34.215536Z","iopub.execute_input":"2022-07-23T05:29:34.215931Z","iopub.status.idle":"2022-07-23T05:29:34.221264Z","shell.execute_reply.started":"2022-07-23T05:29:34.215895Z","shell.execute_reply":"2022-07-23T05:29:34.220490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate the accuray\ndef calculate_accuracy(y_pred, y):\n    top_pred = y_pred.argmax(1, keepdim = True)\n    correct = top_pred.eq(y.view_as(top_pred)).sum()\n    acc = correct.float() / y.shape[0]\n    return acc","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:35.295128Z","iopub.execute_input":"2022-07-23T05:29:35.295759Z","iopub.status.idle":"2022-07-23T05:29:35.301553Z","shell.execute_reply.started":"2022-07-23T05:29:35.295715Z","shell.execute_reply":"2022-07-23T05:29:35.300397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define a train function \ndef train(model, iterator, optimizer, criterion, device):\n    epoch_loss = 0\n    epoch_acc = 0\n    # set the model top train mode\n    model.train()\n    # iterate through the batches\n    for (x, y) in iterator:\n        # deploy data to device\n        x = x.to(device)\n        y = y.to(device)\n        # initialze the optimizer\n        optimizer.zero_grad()\n        # get the ouput\n        y_pred, _ = model(x)\n        # compute loss\n        loss = criterion(y_pred, y)\n        # compute accuracy\n        acc = calculate_accuracy(y_pred, y)\n        # back propagate the loss\n        loss.backward()\n        # update the parameters\n        optimizer.step()\n        # accumulate the loss amongs the batches\n        epoch_loss += loss.item()\n        epoch_acc += acc.item()\n        \n    return epoch_loss / len(iterator), epoch_acc / len(iterator)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:36.086276Z","iopub.execute_input":"2022-07-23T05:29:36.087003Z","iopub.status.idle":"2022-07-23T05:29:36.094614Z","shell.execute_reply.started":"2022-07-23T05:29:36.086958Z","shell.execute_reply":"2022-07-23T05:29:36.093354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate(model, iterator, criterion, device):\n    epoch_loss = 0\n    epoch_acc = 0\n    # set the model to evaluation mode\n    model.eval()\n    # no grad requires\n    with torch.no_grad():\n        \n        for (x, y) in iterator:\n\n            x = x.to(device)\n            y = y.to(device)\n\n            y_pred, _ = model(x)\n            # calculate the loss in eval\n            loss = criterion(y_pred, y)\n\n            acc = calculate_accuracy(y_pred, y)\n\n            epoch_loss += loss.item()\n            epoch_acc += acc.item()\n        \n    return epoch_loss / len(iterator), epoch_acc / len(iterator)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:37.546596Z","iopub.execute_input":"2022-07-23T05:29:37.547824Z","iopub.status.idle":"2022-07-23T05:29:37.555789Z","shell.execute_reply.started":"2022-07-23T05:29:37.547778Z","shell.execute_reply":"2022-07-23T05:29:37.554382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def epoch_time(start_time, end_time):\n    elapsed_time = end_time - start_time\n    elapsed_mins = int(elapsed_time / 60)\n    elapsed_secs = int(elapsed_time - (elapsed_mins * 60))\n    return elapsed_mins, elapsed_secs","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:38.621727Z","iopub.execute_input":"2022-07-23T05:29:38.622824Z","iopub.status.idle":"2022-07-23T05:29:38.628540Z","shell.execute_reply.started":"2022-07-23T05:29:38.622780Z","shell.execute_reply":"2022-07-23T05:29:38.627401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 10\n\nbest_valid_loss = float('inf')\nhistory = {'train': [], 'val': [], 'acc': []}\nfor epoch in range(EPOCHS):\n    # sync the time\n    start_time = time.monotonic()\n    # train and return metric\n    train_loss, train_acc = train(model, train_loader, optimizer, criterion, device)\n    # eval and return metric\n    valid_loss, valid_acc = evaluate(model, val_loader, criterion, device)\n    # save a best model on val set\n    if valid_loss < best_valid_loss:\n        best_valid_loss = valid_loss\n        torch.save(model.state_dict(), 'baseline_model.pt')\n    # log the loss\n    history['train'].append(train_loss)\n    history['val'].append(valid_loss)\n    history['acc'].append(valid_acc)\n    # sync the time\n    end_time = time.monotonic()\n    # calculate the run time\n    epoch_mins, epoch_secs = epoch_time(start_time, end_time)\n    \n    print(f'Epoch: {epoch+1:02} | Epoch Time: {epoch_mins}m {epoch_secs}s')\n    print(f'\\tTrain Loss: {train_loss:.3f} | Train Acc: {train_acc*100:.2f}%')\n    print(f'\\t Val. Loss: {valid_loss:.3f} |  Val. Acc: {valid_acc*100:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:29:39.206632Z","iopub.execute_input":"2022-07-23T05:29:39.207031Z","iopub.status.idle":"2022-07-23T05:31:39.449699Z","shell.execute_reply.started":"2022-07-23T05:29:39.207000Z","shell.execute_reply":"2022-07-23T05:31:39.448412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_accuracies(history):\n    accuracies = history['acc']\n    plt.plot(accuracies, '-x')\n    plt.xlabel('epoch')\n    plt.ylabel('accuracy')\n    plt.title('Accuracy vs. No. of epochs');","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:39.453144Z","iopub.execute_input":"2022-07-23T05:31:39.453508Z","iopub.status.idle":"2022-07-23T05:31:39.458249Z","shell.execute_reply.started":"2022-07-23T05:31:39.453476Z","shell.execute_reply":"2022-07-23T05:31:39.457496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_accuracies(history)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:39.459817Z","iopub.execute_input":"2022-07-23T05:31:39.460453Z","iopub.status.idle":"2022-07-23T05:31:39.604781Z","shell.execute_reply.started":"2022-07-23T05:31:39.460420Z","shell.execute_reply":"2022-07-23T05:31:39.603969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_losses(history):\n    train_losses = history['train']\n    val_losses = history['val']\n    plt.plot(train_losses, '-bx')\n    plt.plot(val_losses, '-rx')\n    plt.xlabel('epoch')\n    plt.ylabel('loss')\n    plt.legend(['Training', 'Validation'])\n    plt.title('Loss vs. No. of epochs');","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:39.607278Z","iopub.execute_input":"2022-07-23T05:31:39.608302Z","iopub.status.idle":"2022-07-23T05:31:39.614124Z","shell.execute_reply.started":"2022-07-23T05:31:39.608244Z","shell.execute_reply":"2022-07-23T05:31:39.612764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_losses(history)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:39.615811Z","iopub.execute_input":"2022-07-23T05:31:39.616521Z","iopub.status.idle":"2022-07-23T05:31:39.824747Z","shell.execute_reply.started":"2022-07-23T05:31:39.616486Z","shell.execute_reply":"2022-07-23T05:31:39.823499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = '../input/msoc/ChineseMNIST/test/test/'\ntest_csv = '../input/msoc/ChineseMNIST/sample_submission.csv'\n# load the test dataset\ntest_dataset = Custom_Dataset(test_dir, test_csv, transform=transform_img)\n# set the batch and create dataLoaders.\nbatch_size=64\ntest_loader = torch.utils.data.DataLoader(test_dataset, batch_size=batch_size, num_workers=0, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:39.827377Z","iopub.execute_input":"2022-07-23T05:31:39.828110Z","iopub.status.idle":"2022-07-23T05:31:39.844764Z","shell.execute_reply.started":"2022-07-23T05:31:39.828061Z","shell.execute_reply":"2022-07-23T05:31:39.843794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make prediction\nmodel.eval()\nans = []\nwith torch.no_grad():  \n    for (x, y) in test_loader:\n        x = x.to(device)\n        y_pred, _ = model(x)\n        _, res = torch.max(y_pred, dim=1)\n        ans.extend(res.detach().cpu().tolist())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:39.846290Z","iopub.execute_input":"2022-07-23T05:31:39.847698Z","iopub.status.idle":"2022-07-23T05:31:50.271230Z","shell.execute_reply.started":"2022-07-23T05:31:39.847631Z","shell.execute_reply":"2022-07-23T05:31:50.269996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save answer\nans_csv = pd.read_csv(test_csv, header=0)\nans_csv['label'] = ans\nans_csv.to_csv('final_ans.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:31:50.272557Z","iopub.execute_input":"2022-07-23T05:31:50.273009Z","iopub.status.idle":"2022-07-23T05:31:50.289778Z","shell.execute_reply.started":"2022-07-23T05:31:50.272962Z","shell.execute_reply":"2022-07-23T05:31:50.288656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}