{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd #For reading csv files.AAAAAA\nimport numpy as np \nimport matplotlib.pyplot as plt #For plotting.\n\nimport PIL.Image as Image #For working with image files.\n\n#Importing torch\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom torch.utils.data import Dataset,DataLoader #For working with data.\n\nfrom torchvision import models,transforms #For pretrained models,image transformations.","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:38.001629Z","iopub.execute_input":"2023-05-10T11:08:38.002506Z","iopub.status.idle":"2023-05-10T11:08:39.646043Z","shell.execute_reply.started":"2023-05-10T11:08:38.002378Z","shell.execute_reply":"2023-05-10T11:08:39.644994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu') #Use GPU if it's available or else use CPU.\nprint(device) #Prints the device we're using.","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:39.64778Z","iopub.execute_input":"2023-05-10T11:08:39.64816Z","iopub.status.idle":"2023-05-10T11:08:39.749326Z","shell.execute_reply.started":"2023-05-10T11:08:39.64812Z","shell.execute_reply":"2023-05-10T11:08:39.748278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/aptos2019-blindness-detection/\"\n\ntrain_df = pd.read_csv(f\"{path}train.csv\")\nprint(f'No.of.training_samples: {len(train_df)}')\n\ntest_df = pd.read_csv(f'{path}test.csv')\nprint(f'No.of.testing_samples: {len(test_df)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:39.751956Z","iopub.execute_input":"2023-05-10T11:08:39.752572Z","iopub.status.idle":"2023-05-10T11:08:39.783997Z","shell.execute_reply.started":"2023-05-10T11:08:39.75251Z","shell.execute_reply":"2023-05-10T11:08:39.782969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Histogram of label counts.\ntrain_df.diagnosis.hist()\nplt.xticks([0,1,2,3,4])\nplt.grid(False)\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:39.785929Z","iopub.execute_input":"2023-05-10T11:08:39.786344Z","iopub.status.idle":"2023-05-10T11:08:40.040109Z","shell.execute_reply.started":"2023-05-10T11:08:39.786303Z","shell.execute_reply":"2023-05-10T11:08:40.039144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#As you can see,the data is imbalanced.\n#So we've to calculate weights for each class,which can be used in calculating loss.\n\nfrom sklearn.utils import class_weight #For calculating weights for each class.\nclass_weights = class_weight.compute_class_weight(class_weight='balanced',classes=np.array([0,1,2,3,4]),y=train_df['diagnosis'].values)\nclass_weights = torch.tensor(class_weights,dtype=torch.float).to(device)\n \nprint(class_weights) #Prints the calculated weights for the classes.","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:40.044978Z","iopub.execute_input":"2023-05-10T11:08:40.047926Z","iopub.status.idle":"2023-05-10T11:08:46.135737Z","shell.execute_reply.started":"2023-05-10T11:08:40.047869Z","shell.execute_reply":"2023-05-10T11:08:46.1338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#For getting a random image from our training set.\nnum = int(np.random.randint(0,len(train_df)-1,(1,))) #Picks a random number.\nsample_image = (f'{path}train_images/{train_df[\"id_code\"][num]}.png')#Image file.\nsample_image = Image.open(sample_image) \nplt.imshow(sample_image)\nplt.axis('off')\nplt.title(f'Class: {train_df[\"diagnosis\"][num]}') #Class of the random image.\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:46.13737Z","iopub.execute_input":"2023-05-10T11:08:46.137843Z","iopub.status.idle":"2023-05-10T11:08:47.243178Z","shell.execute_reply.started":"2023-05-10T11:08:46.137798Z","shell.execute_reply":"2023-05-10T11:08:47.241424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocess the Data","metadata":{}},{"cell_type":"code","source":"class dataset(Dataset): # Inherits from the Dataset class.\n    '''\n    dataset class overloads the __init__, __len__, __getitem__ methods of the Dataset class. \n    \n    Attributes :\n        df:  DataFrame object for the csv file.\n        data_path: Location of the dataset.\n        image_transform: Transformations to apply to the image.\n        train: A boolean indicating whether it is a training_set or not.\n    '''\n    \n    def __init__(self,df,data_path,image_transform=None,train=True): # Constructor.\n        super(Dataset,self).__init__() #Calls the constructor of the Dataset class.\n        self.df = df\n        self.data_path = data_path\n        self.image_transform = image_transform\n        self.train = train\n        \n    def __len__(self):\n        return len(self.df) #Returns the number of samples in the dataset.\n    \n    def __getitem__(self,index):\n        image_id = self.df['id_code'][index]\n        image = Image.open(f'{self.data_path}/{image_id}.png') #Image.\n        if self.image_transform :\n            image = self.image_transform(image) #Applies transformation to the image.\n        \n        if self.train :\n            label = self.df['diagnosis'][index] #Label.\n            return image,label #If train == True, return image & label.\n        \n        else:\n            return image #If train != True, return image.\n            ","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:47.244639Z","iopub.execute_input":"2023-05-10T11:08:47.245061Z","iopub.status.idle":"2023-05-10T11:08:47.255918Z","shell.execute_reply.started":"2023-05-10T11:08:47.245026Z","shell.execute_reply":"2023-05-10T11:08:47.254695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_transform = transforms.Compose([transforms.Resize([512,512]),\n                                      transforms.ToTensor(),\n                                      transforms.Normalize((0.485, 0.456, 0.406), (0.229, 0.224, 0.225))]) #Transformations to apply to the image.\ndata_set = dataset(train_df,f'{path}train_images',image_transform=image_transform)\n\n#Split the data_set so that valid_set contains 0.1 samples of the data_set. \ntrain_set,valid_set = torch.utils.data.random_split(data_set,[3302,360])","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:47.259971Z","iopub.execute_input":"2023-05-10T11:08:47.260917Z","iopub.status.idle":"2023-05-10T11:08:47.275298Z","shell.execute_reply.started":"2023-05-10T11:08:47.260866Z","shell.execute_reply":"2023-05-10T11:08:47.274293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataloader = DataLoader(train_set,batch_size=32,shuffle=True) #DataLoader for train_set.\nvalid_dataloader = DataLoader(valid_set,batch_size=32,shuffle=False) #DataLoader for validation_set.","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:47.277701Z","iopub.execute_input":"2023-05-10T11:08:47.27843Z","iopub.status.idle":"2023-05-10T11:08:47.284544Z","shell.execute_reply.started":"2023-05-10T11:08:47.278382Z","shell.execute_reply":"2023-05-10T11:08:47.28363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build the Model","metadata":{}},{"cell_type":"code","source":"#Since we've less data, we'll use Transfer learning.\nmodel = models.resnet34(pretrained=True) #Downloads the resnet34 model which is pretrained on Imagenet dataset.\n\n#Replace the Final layer of pretrained resnet34 with 4 new layers.\nmodel.fc = nn.Sequential(nn.Linear(512,256),\n                         nn.ReLU(inplace=True),\n                         nn.Linear(256,128),\n                         nn.ReLU(inplace=True),\n                         nn.Linear(128,64),\n                         nn.ReLU(inplace=True),\n                         nn.Linear(64,5),    \n                    )","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:47.286225Z","iopub.execute_input":"2023-05-10T11:08:47.286984Z","iopub.status.idle":"2023-05-10T11:08:48.582484Z","shell.execute_reply.started":"2023-05-10T11:08:47.286935Z","shell.execute_reply":"2023-05-10T11:08:48.581583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.to(device) #Moves the model to the device.","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:48.58402Z","iopub.execute_input":"2023-05-10T11:08:48.584622Z","iopub.status.idle":"2023-05-10T11:08:48.625496Z","shell.execute_reply.started":"2023-05-10T11:08:48.584578Z","shell.execute_reply":"2023-05-10T11:08:48.624584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create functions for Training & Validation","metadata":{}},{"cell_type":"code","source":"def train(dataloader,model,loss_fn,optimizer):\n    '''\n    train function updates the weights of the model based on the\n    loss using the optimizer in order to get a lower loss.\n    \n    Args :\n         dataloader: Iterator for the batches in the data_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         loss_fn: Calculates the discrepancy between the label & the model's predictions.\n         optimizer: Updates the model weights.\n         \n    Returns :\n         Average loss per batch which is calculated by dividing the losses for all the batches\n         with the number of batches.\n    '''\n\n    model.train() #Sets the model for training.\n    \n    total = 0\n    correct = 0\n    running_loss = 0\n    \n    for batch,(x,y) in enumerate(dataloader): #Iterates through the batches.\n        \n        output = model(x.to(device)) #model's predictions.\n        loss   = loss_fn(output,y.to(device)) #loss calculation.\n       \n        running_loss += loss.item()\n        \n        total        += y.size(0)\n        predictions   = output.argmax(dim=1).cpu().detach() #Index for the highest score for all the samples in the batch.\n        correct      += (predictions == y.cpu().detach()).sum().item() #No.of.cases where model's predictions are equal to the label.\n        \n        optimizer.zero_grad() #Gradient values are set to zero.\n        loss.backward() #Calculates the gradients.\n        optimizer.step() #Updates the model weights.\n             \n    \n    avg_loss = running_loss/len(dataloader) # Average loss for a single batch\n    \n    print(f'\\nTraining Loss = {avg_loss:.6f}',end='\\t')\n    print(f'Accuracy on Training set = {100*(correct/total):.6f}% [{correct}/{total}]') #Prints the Accuracy.\n    \n    return avg_loss","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:48.626896Z","iopub.execute_input":"2023-05-10T11:08:48.627286Z","iopub.status.idle":"2023-05-10T11:08:48.637104Z","shell.execute_reply.started":"2023-05-10T11:08:48.627245Z","shell.execute_reply":"2023-05-10T11:08:48.636016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def validate(dataloader,model,loss_fn):\n    '''\n    validate function calculates the average loss per batch and the accuracy of the model's predictions.\n    \n    Args :\n         dataloader: Iterator for the batches in the data_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         loss_fn: Calculates the discrepancy between the label & the model's predictions.\n    \n    Returns :\n         Average loss per batch which is calculated by dividing the losses for all the batches\n         with the number of batches.\n    '''\n    \n    model.eval() #Sets the model for evaluation.\n    \n    total = 0\n    correct = 0\n    running_loss = 0\n    \n    with torch.no_grad(): #No need to calculate the gradients.\n        \n        for x,y in dataloader:\n            \n            output        = model(x.to(device)) #model's output.\n            loss          = loss_fn(output,y.to(device)).item() #loss calculation.\n            running_loss += loss\n            \n            total        += y.size(0)\n            predictions   = output.argmax(dim=1).cpu().detach()\n            correct      += (predictions == y.cpu().detach()).sum().item()\n            \n    avg_loss = running_loss/len(dataloader) #Average loss per batch.      \n    \n    print(f'\\nValidation Loss = {avg_loss:.6f}',end='\\t')\n    print(f'Accuracy on Validation set = {100*(correct/total):.6f}% [{correct}/{total}]') #Prints the Accuracy.\n    \n    return avg_loss","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:48.638987Z","iopub.execute_input":"2023-05-10T11:08:48.639391Z","iopub.status.idle":"2023-05-10T11:08:48.651955Z","shell.execute_reply.started":"2023-05-10T11:08:48.639351Z","shell.execute_reply":"2023-05-10T11:08:48.650832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optimize the Model","metadata":{}},{"cell_type":"code","source":"def optimize(train_dataloader,valid_dataloader,model,loss_fn,optimizer,nb_epochs):\n    '''\n    optimize function calls the train & validate functions for (nb_epochs) times.\n    \n    Args :\n        train_dataloader: DataLoader for the train_set.\n        valid_dataloader: DataLoader for the valid_set.\n        model: Given an input produces an output by multiplying the input with the model weights.\n        loss_fn: Calculates the discrepancy between the label & the model's predictions.\n        optimizer: Updates the model weights.\n        nb_epochs: Number of epochs.\n        \n    Returns :\n        Tuple of lists containing losses for all the epochs.\n    '''\n    #Lists to store losses for all the epochs.\n    train_losses = []\n    valid_losses = []\n\n    for epoch in range(nb_epochs):\n        print(f'\\nEpoch {epoch+1}/{nb_epochs}')\n        print('-------------------------------')\n        train_loss = train(train_dataloader,model,loss_fn,optimizer) #Calls the train function.\n        train_losses.append(train_loss)\n        valid_loss = validate(valid_dataloader,model,loss_fn) #Calls the validate function.\n        valid_losses.append(valid_loss)\n    \n    print('\\nTraining has completed!')\n    \n    return train_losses,valid_losses","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:48.653421Z","iopub.execute_input":"2023-05-10T11:08:48.653902Z","iopub.status.idle":"2023-05-10T11:08:48.666807Z","shell.execute_reply.started":"2023-05-10T11:08:48.653862Z","shell.execute_reply":"2023-05-10T11:08:48.665753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nloss_fn   = nn.CrossEntropyLoss(weight=class_weights) #CrossEntropyLoss with class_weights.\noptimizer = torch.optim.SGD(model.parameters(),lr=0.001) \nnb_epochs = 30\n#Call the optimize function.\ntrain_losses, valid_losses = optimize(train_dataloader,valid_dataloader,model,loss_fn,optimizer,nb_epochs)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T11:08:48.668528Z","iopub.execute_input":"2023-05-10T11:08:48.669048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot the graph of train_losses & valid_losses against nb_epochs.\nepochs = range(nb_epochs)\nplt.plot(epochs, train_losses, 'g', label='Training loss')\nplt.plot(epochs, valid_losses, 'b', label='validation loss')\nplt.title('Training and Validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the model\ntorch.save(model,'CNN_for_DR.pth')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Testing the model","metadata":{}},{"cell_type":"code","source":"test_set = dataset(test_df,f'{path}test_images',image_transform = image_transform,train = False )\n\ntest_dataloader = DataLoader(test_set, batch_size=32, shuffle=False) #DataLoader for test_set.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test(dataloader,model):\n    '''\n    test function predicts the labels given an image batches.\n    \n    Args :\n         dataloader: DataLoader for the test_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         \n    Returns :\n         List of predicted labels.\n    '''\n    \n    model.eval() #Sets the model for evaluation.\n    \n    labels = [] #List to store the predicted labels.\n    \n    with torch.no_grad():\n        \n        for batch,x in enumerate(dataloader):\n            \n            output = model(x.to(device))\n            \n            predictions = output.argmax(dim=1).cpu().detach().tolist() #Predicted labels for an image batch.\n            labels.extend(predictions)\n                \n    print('Testing has completed')\n            \n    return labels                ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = test(test_dataloader,model) #Calls the test function.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have true labels y_true and predicted labels y_pred\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score\n\naccuracy = accuracy_score(y_true, y_pred)\nprecision = precision_score(y_true, y_pred, average='weighted')\nrecall = recall_score(y_true, y_pred, average='weighted')\n\nprint(\"Accuracy:\", accuracy)\nprint(\"Precision:\", precision)\nprint(\"Recall:\", recall)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}