{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# UTMIST Workshop #3: Histopathologic Cancer Detection\nHello UTMIST community!\n\nIn this workshop, we will be building a model that will compete at a Histopathologic Cancer Detection Competition on Kaggle, which can be found here https://www.kaggle.com/competitions/histopathologic-cancer-detection.\nAs stated on their website, our algorithm will identify metastatic cancer in small image patches taken from larger digital pathology scans. The data for this competition is a slightly modified version of the PatchCamelyon (PCam) benchmark dataset.\n\nAuthors: Berke Altiparmak (altiparmak.berke@gmail.com), Lindy Zhai (lindy.zhai@mail.utoronto.ca)\n\nReferences:\nhttps://www.kaggle.com/code/bonhart/pytorch-cnn-from-scratch for the model.\nhttps://www.kaggle.com/code/vithngphm/mini-project-ie0005 for data visualization.","metadata":{}},{"cell_type":"markdown","source":"# Step 1: Importing libraries\n\nWe will import the libraries that we need to build our CNN model. Keep in mind that those libraries are not specific for this project; we are keep using the same libraries, so it is useful for you to get familiar with them.","metadata":{}},{"cell_type":"code","source":"import os\nfrom glob import glob \nimport numpy as np  # for math operations, one of the most commonly used libraries\nimport pandas as pd  # for handling data and data frames, another essential library\nimport cv2  # OpenCV library, which we will use to \"read\" images and transform them\nimport matplotlib.pyplot as plt  # to visualize data\nfrom tqdm import tqdm_notebook,trange  # to see the progress\nimport gc #garbage collection to save RAM\n\n# the output of plotting commands is displayed inline within Jupyter notebook:\n%matplotlib inline  \n\nfrom sklearn.model_selection import train_test_split  # to split train and validation data\n\n# PyTorch libraries to build a Machine Learning Model:\nimport torch \nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision\nimport torchvision.transforms as transforms\nfrom torch.utils.data import TensorDataset, DataLoader, Dataset\n\nimport time","metadata":{"_uuid":"b0b625dddcf44cd8db7690bf380c6cb25f555cbd","execution":{"iopub.status.busy":"2022-11-19T08:40:17.724668Z","iopub.execute_input":"2022-11-19T08:40:17.724979Z","iopub.status.idle":"2022-11-19T08:40:17.738458Z","shell.execute_reply.started":"2022-11-19T08:40:17.724923Z","shell.execute_reply":"2022-11-19T08:40:17.737593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 2: Understanding and Visualizing The Data\n\nIn this section,","metadata":{}},{"cell_type":"code","source":"#set paths to training and test data\npath = \"../input/\" #adapt this path, when running locally\ntrain_path = path + 'train/'\ntest_path = path + 'test/'\nlabels = pd.read_csv(path+\"train_labels.csv\") # read the provided labels\nsubmission = pd.read_csv('../input/sample_submission.csv')\n\n#Splitting data into train and val\ntrain, val = train_test_split(labels, stratify=labels.label, test_size=0.1)\nprint(len(train), len(val))\n\ndf = pd.DataFrame({'path': glob(os.path.join(train_path,'*.tif'))}) # load the filenames\ndf['id'] = df.path.map(lambda x: x.split('/')[3].split(\".\")[0]) # keep only the file names in 'id'\ndf = df.merge(labels, on = \"id\") # merge labels and filepaths\n\ndf.head(10) # print the first ten entries","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:44:14.606518Z","iopub.execute_input":"2022-11-19T08:44:14.606857Z","iopub.status.idle":"2022-11-19T08:44:16.102342Z","shell.execute_reply.started":"2022-11-19T08:44:14.606806Z","shell.execute_reply":"2022-11-19T08:44:16.101553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(N,df):\n    \"\"\" This functions loads N images using the data df\n    \"\"\"\n    # allocate a numpy array for the images (N, 96x96px, 3 channels, values 0 - 255)\n    X = np.zeros([N,96,96,3],dtype=np.uint8) \n    #convert the labels to a numpy array too\n    y = np.squeeze(df.as_matrix(columns=['label']))[0:N]\n    #read images one by one, tdqm notebook displays a progress bar\n    for i, row in tqdm_notebook(df.iterrows(), total=N):\n        if i == N:\n            break\n        X[i] = cv2.imread(row['path'])\n          \n    return X,y","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:40:19.328487Z","iopub.execute_input":"2022-11-19T08:40:19.328942Z","iopub.status.idle":"2022-11-19T08:40:19.334808Z","shell.execute_reply.started":"2022-11-19T08:40:19.328891Z","shell.execute_reply":"2022-11-19T08:40:19.333941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N=1000\nX,y = load_data(N=N,df=df) # load 1000 images","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:49:04.268211Z","iopub.execute_input":"2022-11-19T08:49:04.2686Z","iopub.status.idle":"2022-11-19T08:49:06.117746Z","shell.execute_reply.started":"2022-11-19T08:49:04.268533Z","shell.execute_reply":"2022-11-19T08:49:06.116963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(8, 8), dpi=150)\nnp.random.seed(100) #we can use the seed to get a different set of random images\nimage_count = 16  # number of images we want to have\nfor plot_idx,image_idx in enumerate(np.random.randint(0,N,image_count)):\n    ax = fig.add_subplot(4, image_count//4, plot_idx+1, xticks=[], yticks=[]) #add subplots\n    plt.imshow(X[image_idx]) #plot image\n    ax.set_title('Label: ' + str(y[image_idx])) #show the label corresponding to the image","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:41:11.509616Z","iopub.execute_input":"2022-11-19T08:41:11.509929Z","iopub.status.idle":"2022-11-19T08:41:12.943857Z","shell.execute_reply.started":"2022-11-19T08:41:11.509874Z","shell.execute_reply":"2022-11-19T08:41:12.94308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(4, 2),dpi=100)\nnumber_of_negatives = (y==0).sum()\nnumber_of_positives = (y==1).sum()\nnegative_samples = X[y == 0]\npositive_samples = X[y == 1]\nplt.bar([1,0], [number_of_negatives, number_of_positives]); #plot a bar chart of the label frequency\nplt.xticks([1,0],[\"Negative (N={})\".format(number_of_negatives),\"Positive (N={})\".format(number_of_positives)]);\nplt.ylabel(\"# of samples\")","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:41:16.465005Z","iopub.execute_input":"2022-11-19T08:41:16.465295Z","iopub.status.idle":"2022-11-19T08:41:16.673798Z","shell.execute_reply.started":"2022-11-19T08:41:16.465243Z","shell.execute_reply":"2022-11-19T08:41:16.672676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at the difference in colors of positive samples and negative samples.\n\nFirst, compare each color channel (red, green, blue) of positive and negatives.\n\nThen, make a cumulative comparison.\n\nFinally, compare overall brightness.\n\nScroll back up and compare your conclusions with the 16 image sample you have. Is it accurate? Specifically, check if your observations for red and brightness is generally true so far.","metadata":{}},{"cell_type":"code","source":"nr_of_bins = 256 #each possible pixel value will get a bin in the following histograms\nfig,axs = plt.subplots(4,2,sharey=True,figsize=(8,8),dpi=100)\nrgb_list = [\"Red\", \"Green\", \"Blue\", \"RGB\"]\n\nfor row_idx in range(0, 4):\n    for col_idx in range(0, 2):\n        # do these two below later berke\n        axs[row_idx,0].set_ylabel(\"Relative frequency\")\n        axs[row_idx,1].set_ylabel(rgb_list[row_idx],rotation='horizontal',labelpad=35,fontsize=12)\n        if row_idx < 3:\n            if col_idx == 0:\n                axs[row_idx, col_idx].hist(positive_samples[:,:,:,row_idx].flatten(),bins=nr_of_bins,density=True)\n            elif col_idx == 1:\n                axs[row_idx, col_idx].hist(negative_samples[:,:,:,row_idx].flatten(),bins=nr_of_bins,density=True)\n        else:\n            if col_idx == 0:\n                axs[row_idx, col_idx].hist(positive_samples.flatten(),bins=nr_of_bins,density=True)\n            elif col_idx == 1:\n                axs[row_idx, col_idx].hist(negative_samples.flatten(),bins=nr_of_bins,density=True)\n\n# Additional labeling\naxs[0,0].set_title(\"Positive samples (N =\" + str(positive_samples.shape[0]) + \")\");\naxs[0,1].set_title(\"Negative samples (N =\" + str(negative_samples.shape[0]) + \")\");\naxs[3,0].set_xlabel(\"Pixel value\")\naxs[3,1].set_xlabel(\"Pixel value\")\n\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:41:21.594946Z","iopub.execute_input":"2022-11-19T08:41:21.595235Z","iopub.status.idle":"2022-11-19T08:41:29.746587Z","shell.execute_reply.started":"2022-11-19T08:41:21.595182Z","shell.execute_reply":"2022-11-19T08:41:29.74584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 3: Preprocessing the Data","metadata":{}},{"cell_type":"code","source":"class MyDataset(Dataset):\n    def __init__(self, df_data, data_dir = './', transform=None):\n        super().__init__()\n        self.df = df_data.values\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_name,label = self.df[index]\n        img_path = os.path.join(self.data_dir, img_name+'.tif')\n        image = cv2.imread(img_path)\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"_uuid":"a5f0097544dc230bb794e75a163196f08de0c13c","execution":{"iopub.status.busy":"2022-11-19T08:41:46.835387Z","iopub.execute_input":"2022-11-19T08:41:46.835723Z","iopub.status.idle":"2022-11-19T08:41:46.841291Z","shell.execute_reply.started":"2022-11-19T08:41:46.835667Z","shell.execute_reply":"2022-11-19T08:41:46.840333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans_train = transforms.Compose([transforms.ToPILImage(),\n                                  transforms.Pad(64, padding_mode='reflect'),\n                                  transforms.RandomHorizontalFlip(), \n                                  transforms.RandomVerticalFlip(),\n                                  transforms.RandomRotation(20), \n                                  transforms.ToTensor(),\n                                  transforms.Normalize(mean=[0.5, 0.5, 0.5],std=[0.5, 0.5, 0.5])])\n\ntrans_valid = transforms.Compose([transforms.ToPILImage(),\n                                  transforms.Pad(64, padding_mode='reflect'),\n                                  transforms.ToTensor(),\n                                  transforms.Normalize(mean=[0.5, 0.5, 0.5],std=[0.5, 0.5, 0.5])])\n\ndataset_train = MyDataset(df_data=train, data_dir=train_path, transform=trans_train)\ndataset_valid = MyDataset(df_data=val, data_dir=train_path, transform=trans_valid)\n\nloader_train = DataLoader(dataset = dataset_train, batch_size=batch_size, shuffle=True, num_workers=0)\nloader_valid = DataLoader(dataset = dataset_valid, batch_size=batch_size//2, shuffle=False, num_workers=0)","metadata":{"_uuid":"cbacdb7bcd20d414d115cde200f564934b69f308","execution":{"iopub.status.busy":"2022-11-19T08:41:50.24541Z","iopub.execute_input":"2022-11-19T08:41:50.245741Z","iopub.status.idle":"2022-11-19T08:41:50.268197Z","shell.execute_reply.started":"2022-11-19T08:41:50.245687Z","shell.execute_reply":"2022-11-19T08:41:50.267274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 4: Building the CNN Model!","metadata":{"_uuid":"ac596a6914e7455f1326c01eb532d7b36b972e7f"}},{"cell_type":"code","source":"class SimpleCNN(nn.Module):\n    def __init__(self):\n        # ancestor constructor call\n        super(SimpleCNN, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels=3, out_channels=32, kernel_size=3, padding=2)\n        self.conv2 = nn.Conv2d(in_channels=32, out_channels=64, kernel_size=3, padding=2)\n        self.conv3 = nn.Conv2d(in_channels=64, out_channels=128, kernel_size=3, padding=2)\n#         self.conv4 = nn.Conv2d(in_channels=128, out_channels=256, kernel_size=3, padding=2)\n#         self.conv5 = nn.Conv2d(in_channels=256, out_channels=512, kernel_size=3, padding=2)\n        self.bn1 = nn.BatchNorm2d(32)\n        self.bn2 = nn.BatchNorm2d(64)\n        self.bn3 = nn.BatchNorm2d(128)\n#         self.bn4 = nn.BatchNorm2d(256)\n#         self.bn5 = nn.BatchNorm2d(512)\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.avg = nn.AvgPool2d(8)\n        self.fc = nn.Linear(64*7*7, 2) # !!!\n    def forward(self, x):\n        x = self.pool(F.leaky_relu(self.bn1(self.conv1(x)))) # first convolutional layer then batchnorm, then activation then pooling layer.\n        x = self.pool(F.leaky_relu(self.bn2(self.conv2(x))))\n#         x = self.pool(F.leaky_relu(self.bn3(self.conv3(x))))\n#         x = self.pool(F.leaky_relu(self.bn4(self.conv4(x))))\n#         x = self.pool(F.leaky_relu(self.bn5(self.conv5(x))))\n        x = self.avg(x)\n#         print(x.shape) # lifehack to find out the correct dimension for the Linear Layer\n        x = x.view(x.size(0), -1) # !!!\n        x = self.fc(x)\n        return x","metadata":{"_uuid":"f32c84715ab1c585a56b41fc71c05c42d55bfd4b","execution":{"iopub.status.busy":"2022-11-19T08:41:52.691024Z","iopub.execute_input":"2022-11-19T08:41:52.691309Z","iopub.status.idle":"2022-11-19T08:41:52.698856Z","shell.execute_reply.started":"2022-11-19T08:41:52.691256Z","shell.execute_reply":"2022-11-19T08:41:52.697856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SimpleCNN(nn.Module):\n    def __init__(self):\n        # ancestor constructor call\n        super(SimpleCNN, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels=3, out_channels=32, kernel_size=3, padding=2)\n        self.conv2 = nn.Conv2d(in_channels=32, out_channels=64, kernel_size=3, padding=2)\n        self.conv3 = nn.Conv2d(in_channels=64, out_channels=128, kernel_size=3, padding=2)\n        self.conv4 = nn.Conv2d(in_channels=128, out_channels=256, kernel_size=3, padding=2)\n        self.conv5 = nn.Conv2d(in_channels=256, out_channels=512, kernel_size=3, padding=2)\n        self.conv6 = nn.Conv2d(in_channels=512, out_channels=1024, kernel_size=3, padding=2)\n        self.bn1 = nn.BatchNorm2d(32)\n        self.bn2 = nn.BatchNorm2d(64)\n        self.bn3 = nn.BatchNorm2d(128)\n        self.bn4 = nn.BatchNorm2d(256)\n        self.bn5 = nn.BatchNorm2d(512)\n        self.bn6 = nn.BatchNorm2d(1024)\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.avg = nn.AvgPool2d(8)\n        self.fc = nn.Linear(1024 * 1 * 1, 2) # !!!\n    def forward(self, x):\n        x = self.pool(F.leaky_relu(self.bn1(self.conv1(x)))) # first convolutional layer then batchnorm, then activation then pooling layer.\n        x = self.pool(F.leaky_relu(self.bn2(self.conv2(x))))\n        x = self.pool(F.leaky_relu(self.bn3(self.conv3(x))))\n        x = self.pool(F.leaky_relu(self.bn4(self.conv4(x))))\n        x = self.pool(F.leaky_relu(self.bn5(self.conv5(x))))\n        x = self.pool(F.leaky_relu(self.bn6(self.conv6(x))))\n        x = self.avg(x)\n        #print(x.shape) # lifehack to find out the correct dimension for the Linear Layer\n        x = x.view(x.size(0), -1) # !!!\n        x = self.fc(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:50:26.274681Z","iopub.execute_input":"2022-11-19T08:50:26.275564Z","iopub.status.idle":"2022-11-19T08:50:26.285502Z","shell.execute_reply.started":"2022-11-19T08:50:26.275508Z","shell.execute_reply":"2022-11-19T08:50:26.284284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ngc.collect()\n\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:42:03.501555Z","iopub.execute_input":"2022-11-19T08:42:03.501855Z","iopub.status.idle":"2022-11-19T08:42:03.695564Z","shell.execute_reply.started":"2022-11-19T08:42:03.501797Z","shell.execute_reply":"2022-11-19T08:42:03.692946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Important note:** You may notice that in lines with # !!! there is not very clear 512 \\* 1 \\* 1. This is the dimension of the picture before the FC layers (H x W x C), then you have to calculate it manually (in Keras, for example, .Flatten () does everything for you). However, there is one life hack — just make print (x.shape) in forward () (commented out line). You will see the size (batch_size, C, H, W) - you need to multiply everything except the first (batch_size), this will be the first dimension of Linear (), and it is in C H W that you need to \"expand\" x before feeding to Linear ().","metadata":{"_uuid":"a7ad501f705b587971ced8aec0d36d866af0bed2"}},{"cell_type":"code","source":"# Hyper parameters\nnum_epochs = 8\nnum_classes = 2\nbatch_size = 32\nlearning_rate = 0.002\n\n# Device configuration\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\nprint(device)","metadata":{"_uuid":"67522fc88f034ddb684f9fba0e154c341760d1d1","execution":{"iopub.status.busy":"2022-11-19T08:42:05.304246Z","iopub.execute_input":"2022-11-19T08:42:05.304553Z","iopub.status.idle":"2022-11-19T08:42:05.309856Z","shell.execute_reply.started":"2022-11-19T08:42:05.304492Z","shell.execute_reply":"2022-11-19T08:42:05.308849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():\n    print(\"cuda model\")\n    model = SimpleCNN().cuda()","metadata":{"execution":{"iopub.status.busy":"2022-11-19T08:42:15.378219Z","iopub.execute_input":"2022-11-19T08:42:15.3787Z","iopub.status.idle":"2022-11-19T08:42:15.437572Z","shell.execute_reply.started":"2022-11-19T08:42:15.378449Z","shell.execute_reply":"2022-11-19T08:42:15.436637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loss and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adamax(model.parameters(), lr=learning_rate)","metadata":{"_uuid":"a56a159ba70e958c82e518d718be99bd41c17299","execution":{"iopub.status.busy":"2022-11-19T08:42:18.286003Z","iopub.execute_input":"2022-11-19T08:42:18.286321Z","iopub.status.idle":"2022-11-19T08:42:18.291641Z","shell.execute_reply.started":"2022-11-19T08:42:18.286279Z","shell.execute_reply":"2022-11-19T08:42:18.290593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Train the model\ntotal_num = 1000\n\n# total_step = len(loader_train)\ntotal_step = int(total_num/batch_size)\nif total_num%batch_size != 0:\n    total_step += 1 \nprint('total is ', total_step)\nfor epoch in range(num_epochs):\n#     print(epoch)\n    step, data_count = 0,0\n    start = time.time()\n    for images, labels in iter(loader_train):\n        data_count += len(images)\n#     for i, (images, labels) in enumerate(loader_train):\n#         print(images.shape)\n#         start = time.time()\n        if torch.cuda.is_available():\n#             print('cuda here')\n            images = images.cuda()\n            labels = labels.cuda()\n        \n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        \n        # Backward and optimize\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n        if (step+1) % 10 == 0:\n            end = time.time()\n            print(\"============================== step runtime from start of current epoch {} ====================================\".format(end - start))\n            print ('Epoch [{}/{}], Step [{}/{}], Loss: {:.4f}' \n                   .format(epoch+1, num_epochs, step+1, total_step, loss.item()))\n            \n        step += 1\n        if (data_count > total_num):\n            end = time.time()\n            print(\"============================== EPOCH RUN COMPLETED ====================================\",end - start)\n            print ('Epoch [{}/{}], Step [{}/{}], Loss: {:.4f}' \n                   .format(epoch+1, num_epochs, step+1, total_step, loss.item()))\n            break","metadata":{"_uuid":"dca29061dd5d6a60ba95f16dbaea70e6dc822150","execution":{"iopub.status.busy":"2022-11-19T08:59:17.939654Z","iopub.execute_input":"2022-11-19T08:59:17.940288Z","iopub.status.idle":"2022-11-19T09:00:09.523Z","shell.execute_reply.started":"2022-11-19T08:59:17.940232Z","shell.execute_reply":"2022-11-19T09:00:09.522162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Accuracy Check**","metadata":{"_uuid":"9de7463bf94c03149a7121ce2bb96dc05ecf2d7b"}},{"cell_type":"code","source":"# Test the model\n\ntesting_num = 2000\n\nmodel.eval()  # eval mode (batchnorm uses moving mean/variance instead of mini-batch mean/variance)\nwith torch.no_grad():\n    correct = 0\n    total = 0\n    for images, labels in loader_valid:\n        images = images.to(device)\n        labels = labels.to(device)\n        outputs = model(images)\n        _, predicted = torch.max(outputs.data, 1)\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n        \n        if(total>testing_num):\n            break\n          \n    print('Test Accuracy of the model on the 22003 test images: {} %'.format(100 * correct / total))\n\n# Save the model checkpoint\ntorch.save(model.state_dict(), 'model.ckpt')","metadata":{"_uuid":"998c3c05a3c3fb6fc7f2523dddff50496490d051","execution":{"iopub.status.busy":"2022-11-19T09:07:32.70826Z","iopub.execute_input":"2022-11-19T09:07:32.708573Z","iopub.status.idle":"2022-11-19T09:07:41.425113Z","shell.execute_reply.started":"2022-11-19T09:07:32.708518Z","shell.execute_reply":"2022-11-19T09:07:41.424235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CSV submission**","metadata":{"_uuid":"c1c0709b4a1e7edea538eda5168819a7b949eedf"}},{"cell_type":"code","source":"dataset_valid = MyDataset(df_data=sub, data_dir=test_path, transform=trans_valid)\nloader_test = DataLoader(dataset = dataset_valid, batch_size=32, shuffle=False, num_workers=0)","metadata":{"_uuid":"4d03acf3711d8036703041276749bbaf2ef64e71","execution":{"iopub.status.busy":"2022-11-19T08:40:53.49842Z","iopub.status.idle":"2022-11-19T08:40:53.499351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel.eval()\n\npreds = []\nfor batch_i, (data, target) in enumerate(loader_test):\n    data, target = data.cuda(), target.cuda()\n    output = model(data)\n\n    pr = output[:,1].detach().cpu().numpy()\n    for i in pr:\n        preds.append(i)\nsub.shape, len(preds)\nsub['label'] = preds\nsub.to_csv('s.csv', index=False)","metadata":{"_uuid":"b85e1661e660b45486ab1b46a4d925e18f541b75","execution":{"iopub.status.busy":"2022-11-19T08:40:53.500267Z","iopub.status.idle":"2022-11-19T08:40:53.501048Z"},"trusted":true},"execution_count":null,"outputs":[]}]}