{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n# Imports here 1a\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom PIL import Image\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport torch\nimport torchvision\nfrom torch import nn\nfrom torch import optim\nimport torch.nn.functional as F\nfrom torchvision import datasets, transforms, models\nfrom torch.utils.data import Dataset,DataLoader \nimport cv2\nfrom sklearn.utils import class_weight, shuffle\nimport skimage.io\nfrom skimage.transform import resize\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport PIL\nfrom PIL import Image, ImageOps\nfrom sklearn.utils import class_weight, shuffle\nfrom keras.losses import binary_crossentropy\n#from keras.applications.resnet50 import preprocess_input\nimport keras.backend as K\nimport tensorflow as tf\nfrom sklearn.metrics import f1_score, fbeta_score\nfrom keras.utils import Sequence\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom torch.optim import lr_scheduler\nfrom sklearn import metrics\nfrom sklearn.metrics import confusion_matrix\nimport time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nIMG_SIZE = 224\nNUM_CLASSES = 5\nSEED = 77\nTRAIN_NUM = -1 # use 1000 when you just want to explore new idea, use -1 for full train","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-21T07:59:12.129965Z","iopub.execute_input":"2023-06-21T07:59:12.130713Z","iopub.status.idle":"2023-06-21T07:59:15.018501Z","shell.execute_reply.started":"2023-06-21T07:59:12.130667Z","shell.execute_reply":"2023-06-21T07:59:15.017404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"/kaggle/working/resnet50.h5\n","metadata":{"execution":{"iopub.status.busy":"2023-05-24T04:57:55.761796Z","iopub.execute_input":"2023-05-24T04:57:55.763144Z","iopub.status.idle":"2023-05-24T04:57:56.341553Z","shell.execute_reply.started":"2023-05-24T04:57:55.763096Z","shell.execute_reply":"2023-05-24T04:57:56.339722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#path to the dataset\n\npath1 = \"/kaggle/input/aptos2019-blindness-detection/\"\n\ntrain_df_1 = pd.read_csv(f\"{path1}train.csv\")\ntrain_ds_1 = (f\"{path1}train_images/\")\n#test_df = pd.read_csv(f\"{path}test.csv\")\n#test_ds = (f\"{path}test_images/\")\nprint(f'No.of.training_samples: {len(train_df_1)}')\n#print(f'No.of.testing_samples: {len(test_df)}')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-17T19:24:01.755934Z","iopub.execute_input":"2023-06-17T19:24:01.756922Z","iopub.status.idle":"2023-06-17T19:24:01.771464Z","shell.execute_reply.started":"2023-06-17T19:24:01.756860Z","shell.execute_reply":"2023-06-17T19:24:01.770015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df_1)","metadata":{"execution":{"iopub.status.busy":"2023-06-17T19:27:17.985639Z","iopub.execute_input":"2023-06-17T19:27:17.986077Z","iopub.status.idle":"2023-06-17T19:27:18.001063Z","shell.execute_reply.started":"2023-06-17T19:27:17.986041Z","shell.execute_reply":"2023-06-17T19:27:17.999512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#showing the frequency of each types\nbins=[0,1,2,3,4,5]\nplt.hist(train_df.diagnosis,bins = bins, rwidth=0.5 ,color = \"green\", ec='black',align ='left')\nplt.xticks(bins)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T23:22:06.349548Z","iopub.execute_input":"2023-05-08T23:22:06.350252Z","iopub.status.idle":"2023-05-08T23:22:06.581496Z","shell.execute_reply.started":"2023-05-08T23:22:06.350216Z","shell.execute_reply":"2023-05-08T23:22:06.580531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num = int(np.random.randint(0, len(train_df) -1, (1,)))\nsample_image = Image.open(train_ds + train_df['id_code'][num] +\".png\") \nplt.imshow(sample_image)\nplt.axis('off')\nplt.title(f'Class: {train_df[\"diagnosis\"][num]}') #Class of the random image.\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T23:22:10.177749Z","iopub.execute_input":"2023-05-08T23:22:10.178782Z","iopub.status.idle":"2023-05-08T23:22:10.627206Z","shell.execute_reply.started":"2023-05-08T23:22:10.178728Z","shell.execute_reply":"2023-05-08T23:22:10.626200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the GPU is enabled  or not  2a\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T10:54:14.799361Z","iopub.execute_input":"2023-06-16T10:54:14.799778Z","iopub.status.idle":"2023-06-16T10:54:14.903632Z","shell.execute_reply.started":"2023-06-16T10:54:14.799742Z","shell.execute_reply":"2023-06-16T10:54:14.902276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class datasets(Dataset):\n    def __init__(self, df,dataset, transform = None, train = True):\n        #calling Dataset Constructer\n        super(Dataset, self).__init__()\n        self.df = df\n        self.dataset = dataset\n        self.transform = transform\n        self.train = train\n        \n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        image_id = self.df['id_code'][idx]\n        image = Image.open(self.dataset + image_id + \".png\")\n        \n        #applying trasnforamtion\n        if self.transform:\n            image =  self.transform(image)\n        \n        if self.train:\n            label = self.df['diagnosis'][idx]\n            return image, label\n        else:\n            return image\n            ","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:28:46.439995Z","iopub.execute_input":"2023-06-13T19:28:46.440793Z","iopub.status.idle":"2023-06-13T19:28:46.450425Z","shell.execute_reply.started":"2023-06-13T19:28:46.440750Z","shell.execute_reply":"2023-06-13T19:28:46.449206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: Define your transforms for the training, validation, and testing sets\ndata_transforms = transforms.Compose([transforms.Resize([512,512]),\n                                        transforms.RandomRotation(30),\n                                       transforms.RandomResizedCrop(224),\n                                       transforms.RandomHorizontalFlip(),\n                                       transforms.ToTensor(),\n                                       transforms.Normalize([0.485, 0.456, 0.406],\n                                                            [0.229, 0.224, 0.225])])\ntest_transforms = transforms.Compose([transforms.Resize([512,512]),\n                                     transforms.CenterCrop(224),\n                                     transforms.ToTensor(),\n                                     transforms.Normalize([0.485, 0.456, 0.406],\n                                                            [0.229, 0.224, 0.225])])\n\n\n\n# TODO: Load the datasets with ImageFolder\ntrain_datasets = datasets(train_df, f'{path}train_images/', transform=data_transforms)\ntest_datasets = datasets(test_df, f'{path}test_images/',transform=test_transforms, train = False)\n\n#splitting training and validation dataset from the train_images\ntrain_set,valid_set = torch.utils.data.random_split(train_datasets,[3300,362])\n\n# TODO: Using the image datasets and the trainforms, define the dataloaders\ntrainloaders = DataLoader(train_set, batch_size=64, shuffle = True)\ntestloaders = DataLoader(test_datasets, batch_size = 64, shuffle = True)\nvalidloaders = DataLoader(valid_set, batch_size =64, shuffle = True)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-08T23:22:19.227224Z","iopub.execute_input":"2023-05-08T23:22:19.227759Z","iopub.status.idle":"2023-05-08T23:22:19.287275Z","shell.execute_reply.started":"2023-05-08T23:22:19.227709Z","shell.execute_reply":"2023-05-08T23:22:19.286281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_datasets))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T23:22:20.515697Z","iopub.execute_input":"2023-05-08T23:22:20.516370Z","iopub.status.idle":"2023-05-08T23:22:20.521687Z","shell.execute_reply.started":"2023-05-08T23:22:20.516332Z","shell.execute_reply":"2023-05-08T23:22:20.520584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features, train_labels = next(iter (trainloaders))\nprint (f\"Feature batch shape:{train_features.size()}\")\nprint (f\"Labels batch shape: {train_labels.size()}\")\n#print(train_features.)\nindex=0\nwhile index <5:\n    img = train_features[index]. squeeze ()\n    \n    img = (img.T).detach().numpy()\n    #aprint(img)\n    label = train_labels[index]\n    plt.imshow(img)\n    plt.show()\n    print (f\"Label: {label}\")\n    index = index +1","metadata":{"execution":{"iopub.status.busy":"2023-05-08T23:22:34.217465Z","iopub.execute_input":"2023-05-08T23:22:34.217964Z","iopub.status.idle":"2023-05-08T23:22:48.721199Z","shell.execute_reply.started":"2023-05-08T23:22:34.217922Z","shell.execute_reply":"2023-05-08T23:22:48.720204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Creating a Model**","metadata":{}},{"cell_type":"code","source":"model = models.resnet34(pretrained=True)\n#freezing feature layer\nfor param in model.parameters():\n    param.requires_grad = False\nprint(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T21:30:08.789398Z","iopub.execute_input":"2023-06-08T21:30:08.790025Z","iopub.status.idle":"2023-06-08T21:30:09.847235Z","shell.execute_reply.started":"2023-06-08T21:30:08.789983Z","shell.execute_reply":"2023-06-08T21:30:09.846042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#modifying fc layer\nmodel.fc = nn.Sequential(nn.Linear(512,256),\n                         nn.ReLU(inplace=True),\n                         nn.Dropout(p=0.2),\n                         nn.Linear(256,128),\n                         nn.ReLU(inplace=True),\n                         nn.Dropout(p=0.2),\n                         nn.Linear(128,32), \n                         nn.ReLU(inplace=True),\n                         nn.Dropout(p=0.15),\n                         nn.Linear(32,5), \n                         nn.LogSoftmax(dim=1))\n                    \n    \nprint(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T21:30:12.828774Z","iopub.execute_input":"2023-06-08T21:30:12.829602Z","iopub.status.idle":"2023-06-08T21:30:12.845812Z","shell.execute_reply.started":"2023-06-08T21:30:12.829551Z","shell.execute_reply":"2023-06-08T21:30:12.844502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#moving the model to device (cuda/cpu)\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T21:30:17.805836Z","iopub.execute_input":"2023-06-08T21:30:17.806256Z","iopub.status.idle":"2023-06-08T21:30:22.879166Z","shell.execute_reply.started":"2023-06-08T21:30:17.806219Z","shell.execute_reply":"2023-06-08T21:30:22.878087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loss function\ncriterion = nn.NLLLoss()\n\n#optimizer\n\noptimizer = optim.Adam(model.fc.parameters(), lr=0.05)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:36:46.229463Z","iopub.execute_input":"2023-06-13T19:36:46.230075Z","iopub.status.idle":"2023-06-13T19:36:46.236056Z","shell.execute_reply.started":"2023-06-13T19:36:46.230026Z","shell.execute_reply":"2023-06-13T19:36:46.234983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(dataloader,model,loss_fn,optimizer):\n    '''\n    train function updates the weights of the model based on the\n    loss using the optimizer in order to get a lower loss.\n    \n    Args :\n         dataloader: Iterator for the batches in the data_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         loss_fn: Calculates the discrepancy between the label & the model's predictions.\n         optimizer: Updates the model weights.\n         \n    Returns :\n         Average loss per batch which is calculated by dividing the losses for all the batches\n         with the number of batches.\n    '''\n    print(\"training started\")\n    model.train() #Sets the model for training.\n    \n    total = 0\n    correct = 0\n    running_loss = 0\n    \n    for batch,(x,y) in enumerate(dataloader): #Iterates through the batches.\n        \n        output = model(x.to(device)) #model's predictions.\n        loss   = loss_fn(output,y.to(device)) #loss calculation.\n       \n        running_loss += loss.item()\n        \n        total        += y.size(0)\n        predictions   = output.argmax(dim=1).cpu().detach() #Index for the highest score for all the samples in the batch.\n        correct      += (predictions == y.cpu().detach()).sum().item() #No.of.cases where model's predictions are equal to the label.\n        \n        optimizer.zero_grad() #Gradient values are set to zero.\n        loss.backward() #Calculates the gradients.\n        optimizer.step() #Updates the model weights.\n             \n    \n    avg_loss = running_loss/len(dataloader) # Average loss for a single batch\n    \n    print(f'\\nTraining Loss = {avg_loss:.6f}',end='\\t')\n    print(f'Accuracy on Training set = {100*(correct/total):.6f}% [{correct}/{total}]') #Prints the Accuracy.\n    \n    return avg_loss","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:36:46.794220Z","iopub.execute_input":"2023-06-13T19:36:46.795000Z","iopub.status.idle":"2023-06-13T19:36:46.807703Z","shell.execute_reply.started":"2023-06-13T19:36:46.794954Z","shell.execute_reply":"2023-06-13T19:36:46.806196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def validate(dataloader,model,loss_fn):\n    '''\n    validate function calculates the average loss per batch and the accuracy of the model's predictions.\n    \n    Args :\n         dataloader: Iterator for the batches in the data_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         loss_fn: Calculates the discrepancy between the label & the model's predictions.\n    \n    Returns :\n         Average loss per batch which is calculated by dividing the losses for all the batches\n         with the number of batches.\n    '''\n    \n    model.eval() #Sets the model for evaluation.\n    \n    total = 0\n    correct = 0\n    running_loss = 0\n    \n    with torch.no_grad(): #No need to calculate the gradients.\n        \n        for x,y in dataloader:\n            \n            output        = model(x.to(device)) #model's output.\n            loss          = loss_fn(output,y.to(device)).item() #loss calculation.\n            running_loss += loss\n            \n            total        += y.size(0)\n            predictions   = output.argmax(dim=1).cpu().detach()\n            correct      += (predictions == y.cpu().detach()).sum().item()\n            \n    avg_loss = running_loss/len(dataloader) #Average loss per batch.      \n    \n    print(f'\\nValidation Loss = {avg_loss:.6f}',end='\\t')\n    print(f'Accuracy on Validation set = {100*(correct/total):.6f}% [{correct}/{total}]') #Prints the Accuracy.\n    \n    return avg_loss","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:36:48.006615Z","iopub.execute_input":"2023-06-13T19:36:48.007259Z","iopub.status.idle":"2023-06-13T19:36:48.018160Z","shell.execute_reply.started":"2023-06-13T19:36:48.007215Z","shell.execute_reply":"2023-06-13T19:36:48.016916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def optimize(train_dataloader,valid_dataloader,model,loss_fn,optimizer,nb_epochs):\n    '''\n    optimize function calls the train & validate functions for (nb_epochs) times.\n    \n    Args :\n        train_dataloader: DataLoader for the train_set.\n        valid_dataloader: DataLoader for the valid_set.\n        model: Given an input produces an output by multiplying the input with the model weights.\n        loss_fn: Calculates the discrepancy between the label & the model's predictions.\n        optimizer: Updates the model weights.\n        nb_epochs: Number of epochs.\n        \n    Returns :\n        Tuple of lists containing losses for all the epochs.\n    '''\n    #Lists to store losses for all the epochs.\n    train_losses = []\n    valid_losses = []\n\n    for epoch in range(nb_epochs):\n        print(f'\\nEpoch {epoch+1}/{nb_epochs}')\n        print('-------------------------------')\n        train_loss = train(train_dataloader,model,loss_fn,optimizer) #Calls the train function.\n        print(train_loss)\n        train_losses.append(train_loss)\n        valid_loss = validate(valid_dataloader,model,loss_fn) #Calls the validate function.\n        valid_losses.append(valid_loss)\n    \n    print('\\nTraining has completed!')\n    \n    return train_losses,valid_losses","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:36:49.661280Z","iopub.execute_input":"2023-06-13T19:36:49.662639Z","iopub.status.idle":"2023-06-13T19:36:49.672650Z","shell.execute_reply.started":"2023-06-13T19:36:49.662581Z","shell.execute_reply":"2023-06-13T19:36:49.671231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_epochs = 5\ntrain_losses, valid_losses = optimize(trainloaders,validloaders,model,criterion,optimizer,nb_epochs)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:36:52.041697Z","iopub.execute_input":"2023-06-13T19:36:52.042137Z","iopub.status.idle":"2023-06-13T19:37:08.300886Z","shell.execute_reply.started":"2023-06-13T19:36:52.042098Z","shell.execute_reply":"2023-06-13T19:37:08.299113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot the graph of train_losses & valid_losses against nb_epochs.\nepochs = range(nb_epochs)\nplt.plot(epochs, train_losses, 'g', label='Training loss')\nplt.plot(epochs, valid_losses, 'b', label='validation loss')\nplt.title('Training and Validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-29T10:24:56.337389Z","iopub.execute_input":"2023-03-29T10:24:56.339381Z","iopub.status.idle":"2023-03-29T10:24:56.581997Z","shell.execute_reply.started":"2023-03-29T10:24:56.339340Z","shell.execute_reply":"2023-03-29T10:24:56.580983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test(testloaders, model,criterion):\n    model.eval()\n    acc = 0\n    test_loss = 0\n    with torch.no_grad():        \n        for images, labels in testloaders:\n            \n            images, labels = images.to(device), labels.to(device)\n            \n            logps = model(images)\n            loss = criterion(logps, labels)\n            test_loss += loss.item()\n            \n            #calculate acc\n            ps = torch.exp(logps)\n            top_k, top_class = ps.topk(1, dim=1)\n            equals = top_class == labels.view(*top_class.shape)\n            acc += torch.mean(equals.type(torch.FloatTensor)).item()\n    print( \n      f\"Test accuracy: {acc/len(testloaders):.3f}\"\n      )","metadata":{"execution":{"iopub.status.busy":"2023-03-29T10:24:56.583869Z","iopub.execute_input":"2023-03-29T10:24:56.584605Z","iopub.status.idle":"2023-03-29T10:24:56.592717Z","shell.execute_reply.started":"2023-03-29T10:24:56.584564Z","shell.execute_reply":"2023-03-29T10:24:56.591687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test(testloaders, model,criterion)","metadata":{"execution":{"iopub.status.busy":"2023-03-29T10:24:56.594101Z","iopub.execute_input":"2023-03-29T10:24:56.594644Z","iopub.status.idle":"2023-03-29T10:25:00.637830Z","shell.execute_reply.started":"2023-03-29T10:24:56.594605Z","shell.execute_reply":"2023-03-29T10:25:00.635940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test(dataloader,model):\n    '''\n    test function predicts the labels given an image batches.\n    \n    Args :\n         dataloader: DataLoader for the test_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         \n    Returns :\n         List of predicted labels.\n    '''\n    #correct = 0\n    #total = 0\n    model.eval() #Sets the model for evaluation.\n    \n    labels = [] #List to store the predicted labels.\n    \n    with torch.no_grad():\n        \n        for batch,x in enumerate(dataloader):\n            \n            output = model(x.to(device))\n            \n            predictions = output.argmax(dim=1).cpu().detach().tolist() #Predicted labels for an image batch.\n           # predictions   = output.argmax(dim=1).cpu().detach() #Index for the highest score for all the samples in the batch.\n           # total        += y.size(0)\n            #correct      += (predictions == y.cpu().detach()).sum().item()\n            labels.extend(predictions)\n                \n    print('Testing has completed')\n    #print(f'Accuracy on Training set = {100*(correct/total):.6f}% [{correct}/{total}]')        \n    return labels","metadata":{"execution":{"iopub.status.busy":"2023-03-29T10:26:28.100278Z","iopub.execute_input":"2023-03-29T10:26:28.100827Z","iopub.status.idle":"2023-03-29T10:26:28.113529Z","shell.execute_reply.started":"2023-03-29T10:26:28.100779Z","shell.execute_reply":"2023-03-29T10:26:28.112477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = test(testloaders,model) #Calls the test function.","metadata":{"execution":{"iopub.status.busy":"2023-03-29T10:26:29.302781Z","iopub.execute_input":"2023-03-29T10:26:29.303269Z","iopub.status.idle":"2023-03-29T10:28:31.069001Z","shell.execute_reply.started":"2023-03-29T10:26:29.303225Z","shell.execute_reply":"2023-03-29T10:28:31.067393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels\n","metadata":{"execution":{"iopub.status.busy":"2023-03-29T10:28:35.491044Z","iopub.execute_input":"2023-03-29T10:28:35.492337Z","iopub.status.idle":"2023-03-29T10:28:35.514510Z","shell.execute_reply.started":"2023-03-29T10:28:35.492288Z","shell.execute_reply":"2023-03-29T10:28:35.513145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#path to the dataset 3a\n\npath = \"/kaggle/input/diabetic-retinopathy-resized/\"\n\ntrain_df_2 = pd.read_csv(f\"{path}trainLabels_cropped.csv\")\ntrain_ds_2 = (f\"{path}/resized_train_cropped/resized_train_cropped\")\n#test_df = pd.read_csv(f\"{path}test.csv\")\n#test_ds = (f\"{path}test_images/\")\nprint(f'No.of.training_samples: {len(train_df_2)}')\n#print(f'No.of.testing_samples: {len(test_df)}')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:06:36.964505Z","iopub.execute_input":"2023-06-20T23:06:36.965141Z","iopub.status.idle":"2023-06-20T23:06:37.012051Z","shell.execute_reply.started":"2023-06-20T23:06:36.965100Z","shell.execute_reply":"2023-06-20T23:06:37.010799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_2.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T19:09:55.300638Z","iopub.status.idle":"2023-06-18T19:09:55.301723Z","shell.execute_reply.started":"2023-06-18T19:09:55.301412Z","shell.execute_reply":"2023-06-18T19:09:55.301446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_2.drop(['Unnamed: 0', 'Unnamed: 0.1'],inplace =True, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:06:49.498541Z","iopub.execute_input":"2023-06-20T23:06:49.499327Z","iopub.status.idle":"2023-06-20T23:06:49.509064Z","shell.execute_reply.started":"2023-06-20T23:06:49.499285Z","shell.execute_reply":"2023-06-20T23:06:49.507523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path1 = \"/kaggle/input/aptos2019-blindness-detection/\"\n\ntrain_df_1 = pd.read_csv(f\"{path1}train.csv\")\ntrain_ds_1 = (f\"{path1}train_images/\")\nprint(f'No.of.training_samples: {len(train_df_1)}')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:27:07.576353Z","iopub.execute_input":"2023-06-20T21:27:07.577294Z","iopub.status.idle":"2023-06-20T21:27:07.594602Z","shell.execute_reply.started":"2023-06-20T21:27:07.577255Z","shell.execute_reply":"2023-06-20T21:27:07.593492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_1.rename(columns={'id_code':'image','diagnosis': 'level'}, inplace= True)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:27:09.819800Z","iopub.execute_input":"2023-06-20T21:27:09.820422Z","iopub.status.idle":"2023-06-20T21:27:09.826714Z","shell.execute_reply.started":"2023-06-20T21:27:09.820384Z","shell.execute_reply":"2023-06-20T21:27:09.825525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df_1.info() )\nprint(train_df_2.info())","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:02:57.647155Z","iopub.execute_input":"2023-06-18T16:02:57.649903Z","iopub.status.idle":"2023-06-18T16:02:57.692874Z","shell.execute_reply.started":"2023-06-18T16:02:57.649854Z","shell.execute_reply":"2023-06-18T16:02:57.691815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.concat([train_df_1  ,train_df_2], axis = 0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:27:14.916027Z","iopub.execute_input":"2023-06-20T21:27:14.916547Z","iopub.status.idle":"2023-06-20T21:27:14.926831Z","shell.execute_reply.started":"2023-06-20T21:27:14.916454Z","shell.execute_reply":"2023-06-20T21:27:14.925219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df_2","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:06:53.129949Z","iopub.execute_input":"2023-06-20T23:06:53.130325Z","iopub.status.idle":"2023-06-20T23:06:53.135977Z","shell.execute_reply.started":"2023-06-20T23:06:53.130290Z","shell.execute_reply":"2023-06-20T23:06:53.134678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:02:57.710263Z","iopub.execute_input":"2023-06-18T16:02:57.711696Z","iopub.status.idle":"2023-06-18T16:02:57.737209Z","shell.execute_reply.started":"2023-06-18T16:02:57.711655Z","shell.execute_reply":"2023-06-18T16:02:57.736154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:02:57.741685Z","iopub.execute_input":"2023-06-18T16:02:57.744383Z","iopub.status.idle":"2023-06-18T16:02:57.765467Z","shell.execute_reply.started":"2023-06-18T16:02:57.744344Z","shell.execute_reply":"2023-06-18T16:02:57.764483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**distribution of target variable**","metadata":{}},{"cell_type":"code","source":"train_df['level'].value_counts().plot(kind='pie', autopct='%1.1f%%', startangle=90, colors=['#ae988a', '#F7CAC9', '#EEDD82', '#FFA07A', '#D4D6D5'])\nplt.axis('equal')\nplt.legend(labels=['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR'], loc='upper left', bbox_to_anchor=(-0.1, 1.))\nplt.title('Distribution of Diabetic Retinopathy Levels')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T11:44:37.627717Z","iopub.execute_input":"2023-06-18T11:44:37.628322Z","iopub.status.idle":"2023-06-18T11:44:37.928562Z","shell.execute_reply.started":"2023-06-18T11:44:37.628276Z","shell.execute_reply":"2023-06-18T11:44:37.927342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#4\nx = train_df['image']\ny= train_df['level']\nx, y = shuffle(x, y, random_state=SEED)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:26:48.292112Z","iopub.execute_input":"2023-06-20T21:26:48.292575Z","iopub.status.idle":"2023-06-20T21:26:48.804122Z","shell.execute_reply.started":"2023-06-20T21:26:48.292539Z","shell.execute_reply":"2023-06-20T21:26:48.802519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****to create a class that has equal no of images ****","metadata":{}},{"cell_type":"code","source":"#4a\ntrain_dataset = []\n# Create a dictionary to store the counts of each class\nclass_counts = {}\n\n# Shuffle the training dataset randomly\ntrain_df = train_df.sample(frac=1).reset_index(drop=True)\n\n# Iterate over the training dataset\nfor index, row in train_df.iterrows():\n    image_name = row[\"image\"]\n    class_label = row[\"level\"]\n    \n    image_path = f\"../input/diabetic-retinopathy-resized/resized_train/resized_train/{image_name}.jpeg\"\n    #isExist= os.path.exists(image_path)\n    #if not isExist:\n        #image_path = f\"/kaggle/input/aptos2019-blindness-detection/train_images/{image_name}.png\"\n    #image_path = os.path.join(train_ds, f\"{image_name}.jpeg\")\n    \n    # Check if the count of the class for the image is less than 2000\n    # or if it belongs to class 0 and the count is less than 2500\n    if class_counts.get(class_label, 0) < 7000 or (class_label == 0 and class_counts.get(class_label, 0) < 7000):\n        # Add the image path and class label to the train dataset list\n        train_dataset.append((image_name, class_label))\n        \n        # Increment the count for that class\n        class_counts[class_label] = class_counts.get(class_label, 0) + 1\n\n    # Break the loop if the train dataset contains 2000 images for each class\n    if all(count >= 2000 for count in class_counts.values()):\n        break\n\n# Print the counts of each class in the train dataset\nfor class_label, count in class_counts.items():\n    print(f\"Class {class_label}: {count} images\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:08.159538Z","iopub.execute_input":"2023-06-20T23:07:08.160588Z","iopub.status.idle":"2023-06-20T23:07:10.632125Z","shell.execute_reply.started":"2023-06-20T23:07:08.160527Z","shell.execute_reply":"2023-06-20T23:07:10.630924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T19:12:17.265070Z","iopub.execute_input":"2023-06-18T19:12:17.265606Z","iopub.status.idle":"2023-06-18T19:12:17.281012Z","shell.execute_reply.started":"2023-06-18T19:12:17.265561Z","shell.execute_reply":"2023-06-18T19:12:17.279965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#5a\nx = pd.DataFrame(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:14.858431Z","iopub.execute_input":"2023-06-20T23:07:14.859777Z","iopub.status.idle":"2023-06-20T23:07:14.875494Z","shell.execute_reply.started":"2023-06-20T23:07:14.859720Z","shell.execute_reply":"2023-06-20T23:07:14.874365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:27:41.129297Z","iopub.execute_input":"2023-06-20T21:27:41.129977Z","iopub.status.idle":"2023-06-20T21:27:41.148453Z","shell.execute_reply.started":"2023-06-20T21:27:41.129932Z","shell.execute_reply":"2023-06-20T21:27:41.147332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**splitting the data**","metadata":{}},{"cell_type":"code","source":"#train_x, valid_x, train_y, valid_y = train_test_split(x, y, test_size=0.15,  \n #                                                     stratify=y, random_state=SEED)  5a\n\nX_train, X_test, y_train, y_test = train_test_split(x[0], x[1],\n    test_size=0.10,  random_state = SEED,stratify=x[1])\n\n# Use the same function above for the validation set\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, \n    test_size=0.10,  random_state = SEED,stratify=y_train)\n\nprint(X_train.shape, y_train.shape, X_val.shape, y_val.shape, X_test.shape, y_test.shape)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:17.820228Z","iopub.execute_input":"2023-06-20T23:07:17.820729Z","iopub.status.idle":"2023-06-20T23:07:17.850833Z","shell.execute_reply.started":"2023-06-20T23:07:17.820689Z","shell.execute_reply":"2023-06-20T23:07:17.849824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" y_test\n    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**without splitting equal no of images**","metadata":{}},{"cell_type":"code","source":"#train_x, valid_x, train_y, valid_y = train_test_split(x, y, test_size=0.15,  \n #                                                     stratify=y, random_state=SEED)  5\n\nX_train, X_test, y_train, y_test = train_test_split(x, y,\n    test_size=0.10,  random_state = SEED,stratify=y)\n\n# Use the same function above for the validation set\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, \n    test_size=0.10,  random_state = SEED,stratify=y_train)\n\nprint(X_train.shape, y_train.shape, X_val.shape, y_val.shape, X_test.shape, y_test.shape)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-17T23:45:38.369425Z","iopub.execute_input":"2023-06-17T23:45:38.369848Z","iopub.status.idle":"2023-06-17T23:45:38.408312Z","shell.execute_reply.started":"2023-06-17T23:45:38.369811Z","shell.execute_reply":"2023-06-17T23:45:38.406674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2023-06-05T12:40:29.240004Z","iopub.execute_input":"2023-06-05T12:40:29.241620Z","iopub.status.idle":"2023-06-05T12:40:29.252553Z","shell.execute_reply.started":"2023-06-05T12:40:29.241540Z","shell.execute_reply":"2023-06-05T12:40:29.251180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****","metadata":{}},{"cell_type":"markdown","source":"**Count of each class without**","metadata":{}},{"cell_type":"code","source":"colors = ['red', 'green', 'blue']\nplt.hist([y_train.round(),y_test.round(), y_val.round() ], color=colors)\n#plt.gca().xaxis.set_major_formatter(len(set(y_train)))\nplt.legend(['Train dataset', 'Test dataset', 'Validation dataset'])\nplt.title('count of each class')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T21:28:13.029559Z","iopub.execute_input":"2023-06-20T21:28:13.030590Z","iopub.status.idle":"2023-06-20T21:28:13.418275Z","shell.execute_reply.started":"2023-06-20T21:28:13.030547Z","shell.execute_reply":"2023-06-20T21:28:13.417210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Count of each class with**","metadata":{}},{"cell_type":"code","source":"colors = ['red', 'green', 'blue']\nplt.hist([y_train.round(),y_test.round(), y_val.round() ], color=colors)\n#plt.gca().xaxis.set_major_formatter(len(set(y_train)))\nplt.legend(['Train dataset', 'Test dataset', 'Validation dataset'])\nplt.title('count of each class')","metadata":{"execution":{"iopub.status.busy":"2023-06-17T23:46:00.735893Z","iopub.execute_input":"2023-06-17T23:46:00.736335Z","iopub.status.idle":"2023-06-17T23:46:01.104610Z","shell.execute_reply.started":"2023-06-17T23:46:00.736292Z","shell.execute_reply":"2023-06-17T23:46:01.103309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preprocessing from kaggle\nimage_size = 256\n#IMAGE PREPROCESSING\n\ndef prepare_image(path, \n                  sigmaX         = 10, \n                  do_random_crop = False):\n    \n    '''\n    Preprocess image\n    '''\n    \n    # import image\n    image = cv2.imread(path)\n    #image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # perform smart crops\n    image = crop_black(image, tol = 7)\n    if do_random_crop == True:\n        image = random_crop(image, size = (0.9, 1))\n    \n    # resize and color\n    image = cv2.resize(image, (int(image_size), int(image_size)))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), sigmaX), -4, 128)\n    \n    # circular crop\n    image = circle_crop(image, sigmaX = sigmaX)\n\n    # convert to tensor    \n    image = torch.tensor(image)\n    image = image.permute(2, 1, 0)\n    return image\n\n#CROP FUNCTIONS\n\ndef crop_black(img, \n               tol = 7):\n    \n    '''\n    Perform automatic crop of black areas\n    '''\n    \n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    \n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img > tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        \n        if (check_shape == 0): \n            return img \n        else:\n            img1 = img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2 = img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3 = img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img  = np.stack([img1, img2, img3], axis = -1)\n            return img\ndef circle_crop(img, \n                sigmaX = 10):   \n    \n    '''\n    Perform circular crop around image center\n    '''\n        \n    height, width, depth = img.shape\n    \n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n    \n    x = int(width / 2)\n    y = int(height / 2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness = -1)\n    \n    img = cv2.bitwise_and(img, img, mask = circle_img)\n    return img \ndef random_crop(img, \n                size = (0.9, 1)):\n    \n    '''\n    Random crop\n    '''\n\n    height, width, depth = img.shape\n    \n    cut = 1 - random.uniform(size[0], size[1])\n    \n    i = random.randint(0, int(cut * height))\n    j = random.randint(0, int(cut * width))\n    h = i + int((1 - cut) * height)\n    w = j + int((1 - cut) * width)\n\n    img = img[i:h, j:w, :]    \n    \n    return img","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:35:33.922820Z","iopub.execute_input":"2023-06-21T07:35:33.923832Z","iopub.status.idle":"2023-06-21T07:35:33.948382Z","shell.execute_reply.started":"2023-06-21T07:35:33.923783Z","shell.execute_reply":"2023-06-21T07:35:33.947227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **Eye Preprocessing - more clarity of an image using this processing techniques by ben 'https://www.kaggle.com/code/ratthachat/aptos-eye-preprocessing-in-diabetic-retinopathy/notebook'**","metadata":{}},{"cell_type":"code","source":"# 6a\ndef load_ben_color(path, sigmaX=10):\n    #print(path)\n    image = cv2.imread(path)\n    if image is None or image.size == 0:\n        raise ValueError(\"Failed to load or empty image: \" + path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = crop_image_from_gray(image)\n    image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n    image=cv2.addWeighted ( image,4, cv2.GaussianBlur( image , (0,0) , sigmaX) ,-4 ,128)\n        \n    return image","metadata":{"execution":{"iopub.status.busy":"2023-06-19T01:36:39.335557Z","iopub.execute_input":"2023-06-19T01:36:39.336678Z","iopub.status.idle":"2023-06-19T01:36:39.345897Z","shell.execute_reply.started":"2023-06-19T01:36:39.336624Z","shell.execute_reply":"2023-06-19T01:36:39.344621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 7a\ndef crop_image1(img,tol=7):\n    # img is image data\n    # tol  is tolerance\n        \n    mask = img>tol\n    return img[np.ix_(mask.any(1),mask.any(0))]\n\ndef crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n    #         print(img.shape)\n        return img","metadata":{"execution":{"iopub.status.busy":"2023-06-19T01:36:40.378037Z","iopub.execute_input":"2023-06-19T01:36:40.379029Z","iopub.status.idle":"2023-06-19T01:36:40.391699Z","shell.execute_reply.started":"2023-06-19T01:36:40.378977Z","shell.execute_reply":"2023-06-19T01:36:40.390514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"NUM_SAMP=10\nfig = plt.figure(figsize=(25, 16))\nfor class_id in sorted(y_train.unique()):\n    for i, (idx, row) in enumerate(train_df.loc[train_df['level'] == class_id].sample(NUM_SAMP, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, NUM_SAMP, class_id * NUM_SAMP + i + 1, xticks=[], yticks=[])\n        path=f\"../input/diabetic-retinopathy-resized/resized_train/resized_train/{row['image']}.jpeg\"\n        image = prepare_image(path,do_random_crop = False)\n\n        plt.imshow(image)\n        ax.set_title('%d-%d-%s' % (class_id, idx, row['image']) )","metadata":{"execution":{"iopub.status.busy":"2023-06-19T01:41:50.601705Z","iopub.execute_input":"2023-06-19T01:41:50.602651Z","iopub.status.idle":"2023-06-19T01:41:56.155421Z","shell.execute_reply.started":"2023-06-19T01:41:50.602588Z","shell.execute_reply":"2023-06-19T01:41:56.153948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Comparision**","metadata":{}},{"cell_type":"code","source":"NUM_SAMP=10\nfig = plt.figure(figsize=(25, 16))\nfor class_id in sorted(y_train.unique()):\n    for i, (idx, row) in enumerate(train_df.loc[train_df['level'] == class_id].sample(NUM_SAMP, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, NUM_SAMP, class_id * NUM_SAMP + i + 1, xticks=[], yticks=[])\n        path=f\"../input/diabetic-retinopathy-resized/resized_train/resized_train/{row['image']}.jpeg\"\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n#         image = crop_image_from_gray(image)\n        image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n#         image=cv2.addWeighted ( image,4, cv2.GaussianBlur( image , (0,0) , IMG_SIZE/10) ,-4 ,128)\n\n        plt.imshow(image, cmap='gray')\n        ax.set_title('%d-%d-%s' % (class_id, idx, row['image']) )","metadata":{"execution":{"iopub.status.busy":"2023-06-19T01:38:56.586539Z","iopub.execute_input":"2023-06-19T01:38:56.587334Z","iopub.status.idle":"2023-06-19T01:39:01.206039Z","shell.execute_reply.started":"2023-06-19T01:38:56.587279Z","shell.execute_reply":"2023-06-19T01:39:01.205027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:39:43.358314Z","iopub.execute_input":"2023-05-22T11:39:43.358788Z","iopub.status.idle":"2023-05-22T11:39:43.826651Z","shell.execute_reply.started":"2023-05-22T11:39:43.358715Z","shell.execute_reply":"2023-05-22T11:39:43.825456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Converting nd array to PIL image**","metadata":{}},{"cell_type":"code","source":"#8a\nclass retinaDataset(Dataset):\n    def __init__(self, file_names, transform=None):\n        self.file_names = file_names\n        self.transform = transform\n        self.df = pd.read_csv(\"../input/diabetic-retinopathy-resized/trainLabels.csv\")\n        \n    def __len__(self):\n        return len(self.file_names)\n    \n    def __getitem__(self, index):\n        #print(index)\n        #print(self.file_names)\n        file_name = self.file_names[index]\n        #print(file_name)\n        \n        image_path = f\"../input/diabetic-retinopathy-resized/resized_train/resized_train/{file_name}.jpeg\"\n        #image = Image.open(image_path)\n        #image = load_ben_color(image_path,sigmaX=30)\n\n        image    = prepare_image(image_path, do_random_crop = False)\n       \n        \n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, torch.tensor(self.df.loc[self.df['image'] == file_name, 'level'].values[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-19T01:43:44.623966Z","iopub.execute_input":"2023-06-19T01:43:44.624809Z","iopub.status.idle":"2023-06-19T01:43:44.636034Z","shell.execute_reply.started":"2023-06-19T01:43:44.624770Z","shell.execute_reply":"2023-06-19T01:43:44.634887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#8 combining 2 dataset\nclass retinaDataset(Dataset):\n    def __init__(self, file_names, train_df, transform=None):\n        self.file_names = file_names\n        self.transform = transform\n        self.df = train_df\n        \n    def __len__(self):\n        return len(self.file_names)\n    \n    def __getitem__(self, index):\n        #print(index)\n        #print(self.file_names)\n        file_name = self.file_names[index]\n        #print(file_name)\n        \n        image_path = f\"../input/diabetic-retinopathy-resized/resized_train/resized_train/{file_name}.jpeg\"\n        #isExist= os.path.exists(image_path)\n        #if not isExist:\n         #   image_path = f\"/kaggle/input/aptos2019-blindness-detection/train_images/{file_name}.png\"\n            \n        #image = Image.open(image_path)\n        #image = load_ben_color(image_path,sigmaX=30)\n\n        image    = prepare_image(image_path, do_random_crop = False)\n       \n        \n        \n        if self.transform:\n            image = self.transform(image)\n        \n        return image, torch.tensor(self.df.loc[self.df['image'] == file_name, 'level'].values[0])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:46.040803Z","iopub.execute_input":"2023-06-20T23:07:46.041855Z","iopub.status.idle":"2023-06-20T23:07:46.051579Z","shell.execute_reply.started":"2023-06-20T23:07:46.041803Z","shell.execute_reply":"2023-06-20T23:07:46.050436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#training transformations 9a\ndata_transforms = transforms.Compose([transforms.ToPILImage(),\n                                      transforms.Resize((299,299)),\n                                        \n                                        transforms.RandomRotation((-360, 360)),\n                                      #transforms.RandomHorizontalFlip(),\n                                     #transforms.RandomVerticalFlip(),\n                                       \n                                       transforms.ToTensor(),])\n                                       #transforms.Normalize([0.485, 0.456, 0.406],\n                                                           # [0.229, 0.224, 0.225])])\nvalid_trans = transforms.Compose([transforms.ToPILImage(),\n                                  transforms.Resize((299,299)),\n                                     \n                                    transforms.ToTensor(),])\n                                     #transforms.Normalize([0.485, 0.456, 0.406],\n                                                        #  [0.229, 0.224, 0.225])])\n#\n# test transformations\ntest_trans = valid_trans\n                                  \n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:48.613784Z","iopub.execute_input":"2023-06-20T23:07:48.614173Z","iopub.status.idle":"2023-06-20T23:07:48.623259Z","shell.execute_reply.started":"2023-06-20T23:07:48.614139Z","shell.execute_reply":"2023-06-20T23:07:48.622149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#resetting index for proper indexing  10a\nX_train = X_train.reset_index()\nX_val = X_val.reset_index()\nX_test = X_test.reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:50.447796Z","iopub.execute_input":"2023-06-20T23:07:50.448999Z","iopub.status.idle":"2023-06-20T23:07:50.459599Z","shell.execute_reply.started":"2023-06-20T23:07:50.448958Z","shell.execute_reply":"2023-06-20T23:07:50.458232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2023-06-18T19:13:25.681659Z","iopub.execute_input":"2023-06-18T19:13:25.682126Z","iopub.status.idle":"2023-06-18T19:13:25.699074Z","shell.execute_reply.started":"2023-06-18T19:13:25.682064Z","shell.execute_reply":"2023-06-18T19:13:25.698134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_rows = 4000\npd.set_option('display.max_colwidth', None)\nX_train[0]","metadata":{"execution":{"iopub.status.busy":"2023-06-05T12:41:12.092906Z","iopub.execute_input":"2023-06-05T12:41:12.093348Z","iopub.status.idle":"2023-06-05T12:41:12.104544Z","shell.execute_reply.started":"2023-06-05T12:41:12.093312Z","shell.execute_reply":"2023-06-05T12:41:12.103338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the datasets \ntrain_dataset = retinaDataset(X_train['image'],train_df, transform=data_transforms)\ntest_dataset = retinaDataset(X_test['image'],train_df, transform=test_trans)\nval_dataset = retinaDataset(X_val['image'],train_df,transform=valid_trans)","metadata":{"execution":{"iopub.status.busy":"2023-06-17T19:46:30.809004Z","iopub.execute_input":"2023-06-17T19:46:30.809674Z","iopub.status.idle":"2023-06-17T19:46:30.823561Z","shell.execute_reply.started":"2023-06-17T19:46:30.809621Z","shell.execute_reply":"2023-06-17T19:46:30.822087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class with equal 11a\ntrain_dataset = retinaDataset(X_train[0],train_df, transform=data_transforms)\ntest_dataset = retinaDataset(X_test[0],train_df, transform=test_trans)\nval_dataset = retinaDataset(X_val[0],train_df, transform=valid_trans)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:53.611428Z","iopub.execute_input":"2023-06-20T23:07:53.612146Z","iopub.status.idle":"2023-06-20T23:07:53.620702Z","shell.execute_reply.started":"2023-06-20T23:07:53.612108Z","shell.execute_reply":"2023-06-20T23:07:53.619541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataloaders 12a\ntrainloaders = DataLoader(train_dataset, batch_size=10, shuffle = True)\ntestloaders = DataLoader(test_dataset, batch_size = 10, shuffle = False)\nvalidloaders = DataLoader(val_dataset, batch_size =10, shuffle = False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:55.255561Z","iopub.execute_input":"2023-06-20T23:07:55.256731Z","iopub.status.idle":"2023-06-20T23:07:55.263240Z","shell.execute_reply.started":"2023-06-20T23:07:55.256669Z","shell.execute_reply":"2023-06-20T23:07:55.262054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#it \ntrain_features, train_labels = next(iter(trainloaders))\nprint (f\"Feature batch shape:{train_features.size()}\")\nprint (f\"Labels batch shape: {train_labels.size()}\")\n#print(train_features.)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:07:55.839227Z","iopub.execute_input":"2023-06-20T23:07:55.839943Z","iopub.status.idle":"2023-06-20T23:07:56.329411Z","shell.execute_reply.started":"2023-06-20T23:07:55.839902Z","shell.execute_reply":"2023-06-20T23:07:56.328286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#printing sample transformation images\nindex=0\nwhile index <5:\n    img = train_features[index]. squeeze ()\n    \n    img = (img.T).detach().numpy()\n    #img = Image.fromarray(img.astype('uint8'), 'RGB')\n    #aprint(img)\n    label = train_labels[index]\n    plt.imshow(img)\n    plt.show()\n    print(train_features[index])\n    print (f\"Label: {label}\")\n    index = index +1","metadata":{"execution":{"iopub.status.busy":"2023-06-20T22:02:03.465890Z","iopub.execute_input":"2023-06-20T22:02:03.466288Z","iopub.status.idle":"2023-06-20T22:02:04.956012Z","shell.execute_reply.started":"2023-06-20T22:02:03.466251Z","shell.execute_reply":"2023-06-20T22:02:04.954972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 13a\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = models.vgg19(pretrained=True)\n#freezing feature layer\nfor param in model.parameters():\n    param.requires_grad = False\n    \n#modifying fc layer\nmodel.classifier = nn.Sequential(nn.Linear(25088, 4096),\n                        nn.ReLU(),\n                        nn.Dropout(p=0.2),\n                        nn.Linear(4096, 256),\n                        nn.ReLU(),\n                        nn.Dropout(p=0.2),\n                        nn.Linear(256, 5),\n                        nn.LogSoftmax(dim=1))\n#moving the model to device (cuda/cpu)\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:34:19.729579Z","iopub.execute_input":"2023-06-13T19:34:19.730412Z","iopub.status.idle":"2023-06-13T19:34:34.032129Z","shell.execute_reply.started":"2023-06-13T19:34:19.730370Z","shell.execute_reply":"2023-06-13T19:34:34.030338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 13 #inceptionv3\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = models.inception_v3(pretrained=True)\n#freezing feature layer\nfor param in model.parameters():\n    param.requires_grad = False\n    \n#modifying fc layer\nmodel.fc = torch.nn.Linear(in_features=2048, out_features=5, bias=True)\nmodel.aux_logits = False\n#moving the model to device (cuda/cpu)\nmodel =model.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:08:05.055512Z","iopub.execute_input":"2023-06-20T23:08:05.055964Z","iopub.status.idle":"2023-06-20T23:08:06.429408Z","shell.execute_reply.started":"2023-06-20T23:08:05.055927Z","shell.execute_reply":"2023-06-20T23:08:06.428385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#modifying fc layer 13a\nmodel.fc = nn.Sequential(nn.Linear(2048,256),\n                         nn.ReLU(inplace=True),\n                         nn.Dropout(p=0.2),\n                         nn.Linear(256,128),\n                         nn.ReLU(inplace=True),\n                         nn.Dropout(p=0.2),\n                         nn.Linear(128,32), \n                         nn.ReLU(inplace=True),\n                         nn.Dropout(p=0.15),\n                         nn.Linear(32,5), \n                         nn.LogSoftmax(dim=1))\n                    \n    \nprint(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:34:34.040913Z","iopub.execute_input":"2023-06-13T19:34:34.045338Z","iopub.status.idle":"2023-06-13T19:34:34.075081Z","shell.execute_reply.started":"2023-06-13T19:34:34.045283Z","shell.execute_reply":"2023-06-13T19:34:34.073920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#moving the model to device (cuda/cpu) 14a\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T00:19:50.718696Z","iopub.execute_input":"2023-06-15T00:19:50.719276Z","iopub.status.idle":"2023-06-15T00:19:50.748369Z","shell.execute_reply.started":"2023-06-15T00:19:50.719224Z","shell.execute_reply":"2023-06-15T00:19:50.744907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #loss function 15a\ncriterion = nn.NLLLoss()\n\n#optimizer\n\noptimizer = optim.Adam(model.fc.parameters(), lr=0.01)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T19:34:44.322525Z","iopub.execute_input":"2023-06-13T19:34:44.323592Z","iopub.status.idle":"2023-06-13T19:34:44.329831Z","shell.execute_reply.started":"2023-06-13T19:34:44.323524Z","shell.execute_reply":"2023-06-13T19:34:44.328474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loss function\ncriterion = nn.NLLLoss()\n\n#optimizer\n\noptimizer = optim.Adam(model.classifier.parameters(), lr=0.0025)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T12:16:46.570760Z","iopub.execute_input":"2023-06-09T12:16:46.571459Z","iopub.status.idle":"2023-06-09T12:16:46.577419Z","shell.execute_reply.started":"2023-06-09T12:16:46.571417Z","shell.execute_reply":"2023-06-09T12:16:46.575929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#inception v3\noptimizer = torch.optim.Adam(model.fc.parameters(), lr = learning_rate)\ncriterion = torch.nn.CrossEntropyLoss()","metadata":{"execution":{"iopub.status.busy":"2023-06-18T19:14:48.624979Z","iopub.execute_input":"2023-06-18T19:14:48.626027Z","iopub.status.idle":"2023-06-18T19:14:48.633910Z","shell.execute_reply.started":"2023-06-18T19:14:48.625964Z","shell.execute_reply":"2023-06-18T19:14:48.631285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = torch.tensor([0.0645, 0.1613, 0.0722, 0.4254, 0.4526])\n\n# Create the loss function with class weights\ncriterion = torch.nn.CrossEntropyLoss(weight=class_weights.to(device), reduction='mean')\nclass WeightedCrossEntropyLoss(torch.nn.Module):\n    def __init__(self, weight=None, reduction='mean'):\n        super(WeightedCrossEntropyLoss, self).__init__()\n        self.weight = weight\n        self.reduction = reduction\n\n    def forward(self, input, target):\n        return F.cross_entropy(input, target, weight=self.weight, reduction=self.reduction)\n\n# Create the custom loss function with class weights\n#criterion = WeightedCrossEntropyLoss(weight=class_weights)\n\noptimizer = torch.optim.Adam(model.fc.parameters(), lr=learning_rate)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T01:04:57.321915Z","iopub.execute_input":"2023-06-21T01:04:57.322581Z","iopub.status.idle":"2023-06-21T01:04:57.332679Z","shell.execute_reply.started":"2023-06-21T01:04:57.322540Z","shell.execute_reply":"2023-06-21T01:04:57.331535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# 16a\ndef train(dataloader,model,loss_fn,optimizer, weight):\n    '''\n    train function updates the weights of the model based on the\n    loss using the optimizer in order to get a lower loss.\n    \n    Args :\n         dataloader: Iterator for the batches in the data_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         loss_fn: Calculates the discrepancy between the label & the model's predictions.\n         optimizer: Updates the model weights.\n         \n    Returns :\n         Average loss per batch which is calculated by dividing the losses for all the batches\n         with the number of batches.\n    '''\n    print(\"training started\")\n    model.train() #Sets the model for training.\n    \n    total = 0\n    correct = 0\n    running_loss = 0\n    \n    for batch,(x,y) in enumerate(dataloader): #Iterates through the batches.\n        x = x.to(device)  # Move input data to the same device as the model\n        y = y.to(device)  # Move input data to the same device as the model\n\n        # ...\n\n        output = model(x)\n        loss = loss_fn(output, y)  \n        #output = model(x.to(device)) #model's predictions.\n        #loss   = loss_fn(output,y.to(device)) #loss calculation.\n       \n        running_loss += loss.item()\n        \n        total        += y.size(0)\n        predictions   = output.argmax(dim=1).cpu().detach() #Index for the highest score for all the samples in the batch.\n        correct      += (predictions == y.cpu().detach()).sum().item() #No.of.cases where model's predictions are equal to the label.\n        \n        optimizer.zero_grad() #Gradient values are set to zero.\n        loss.backward() #Calculates the gradients.\n        optimizer.step() #Updates the model weights.\n             \n    \n    avg_loss = running_loss/len(dataloader) # Average loss for a single batch\n    \n    print(f'\\nTraining Loss = {avg_loss:.6f}',end='\\t')\n    print(f'Accuracy on Training set = {100*(correct/total):.6f}% [{correct}/{total}]') #Prints the Accuracy.\n    \n    return avg_loss","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:08:14.379031Z","iopub.execute_input":"2023-06-20T23:08:14.379419Z","iopub.status.idle":"2023-06-20T23:08:14.392129Z","shell.execute_reply.started":"2023-06-20T23:08:14.379384Z","shell.execute_reply":"2023-06-20T23:08:14.390726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#17a\ndef validate(dataloader,model,loss_fn):\n    '''\n    validate function calculates the average loss per batch and the accuracy of the model's predictions.\n    \n    Args :\n         dataloader: Iterator for the batches in the data_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         loss_fn: Calculates the discrepancy between the label & the model's predictions.\n    \n    Returns :\n         Average loss per batch which is calculated by dividing the losses for all the batches\n         with the number of batches.\n    '''\n    \n    model.eval() #Sets the model for evaluation.\n    \n    total = 0\n    correct = 0\n    running_loss = 0\n    \n    with torch.no_grad(): #No need to calculate the gradients.\n        \n        for x,y in dataloader:\n            \n            output        = model(x.to(device)) #model's output.\n            loss          = loss_fn(output,y.to(device)).item() #loss calculation.\n            running_loss += loss\n            \n            total        += y.size(0)\n            predictions   = output.argmax(dim=1).cpu().detach()\n            correct      += (predictions == y.cpu().detach()).sum().item()\n            \n    avg_loss = running_loss/len(dataloader) #Average loss per batch.      \n    \n    print(f'\\nValidation Loss = {avg_loss:.6f}',end='\\t')\n    print(f'Accuracy on Validation set = {100*(correct/total):.6f}% [{correct}/{total}]') #Prints the Accuracy.\n    \n    return avg_loss","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:08:15.096114Z","iopub.execute_input":"2023-06-20T23:08:15.096961Z","iopub.status.idle":"2023-06-20T23:08:15.108431Z","shell.execute_reply.started":"2023-06-20T23:08:15.096914Z","shell.execute_reply":"2023-06-20T23:08:15.107270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#18a\ndef optimize(train_dataloader,valid_dataloader,model,loss_fn,optimizer,nb_epochs,weight):\n    '''\n    optimize function calls the train & validate functions for (nb_epochs) times.\n    \n    Args :\n        train_dataloader: DataLoader for the train_set.\n        valid_dataloader: DataLoader for the valid_set.\n        model: Given an input produces an output by multiplying the input with the model weights.\n        loss_fn: Calculates the discrepancy between the label & the model's predictions.\n        optimizer: Updates the model weights.\n        nb_epochs: Number of epochs.\n        \n    Returns :\n        Tuple of lists containing losses for all the epochs.\n    '''\n    #Lists to store losses for all the epochs.\n    train_losses = []\n    valid_losses = []\n\n    \n    for epoch in range(nb_epochs):\n        print(f'\\nEpoch {epoch+1}/{nb_epochs}')\n        print('-------------------------------')\n        train_loss = train(train_dataloader,model,loss_fn,optimizer,weight) #Calls the train function.\n        print(train_loss)\n        train_losses.append(train_loss)\n        valid_loss = validate(valid_dataloader,model,loss_fn) #Calls the validate function.\n        valid_losses.append(valid_loss)\n    \n    print('\\nTraining has completed!')\n    \n    return train_losses,valid_losses","metadata":{"execution":{"iopub.status.busy":"2023-06-20T23:08:15.649985Z","iopub.execute_input":"2023-06-20T23:08:15.650369Z","iopub.status.idle":"2023-06-20T23:08:15.660429Z","shell.execute_reply.started":"2023-06-20T23:08:15.650333Z","shell.execute_reply":"2023-06-20T23:08:15.659144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def optimize(train_dataloader, valid_dataloader, model, loss_fn, optimizer, nb_epochs, patience=3):\n    '''\n    optimize function calls the train & validate functions for (nb_epochs) times and includes early stopping.\n    \n    Args :\n        train_dataloader: DataLoader for the train_set.\n        valid_dataloader: DataLoader for the valid_set.\n        model: Given an input produces an output by multiplying the input with the model weights.\n        loss_fn: Calculates the discrepancy between the label & the model's predictions.\n        optimizer: Updates the model weights.\n        nb_epochs: Number of epochs.\n        patience: Number of epochs to wait for improvement before early stopping.\n        \n    Returns :\n        Tuple of lists containing losses for all the epochs.\n    '''\n    train_losses = []\n    valid_losses = []\n    best_loss = float('inf')\n    epochs_without_improvement = 0\n\n    for epoch in range(nb_epochs):\n        print(f'\\nEpoch {epoch+1}/{nb_epochs}')\n        print('-------------------------------')\n        \n        train_loss = train(train_dataloader, model, loss_fn, optimizer)\n        train_losses.append(train_loss)\n        \n        valid_loss = validate(valid_dataloader, model, loss_fn)\n        valid_losses.append(valid_loss)\n        \n        if valid_loss < best_loss:\n            best_loss = valid_loss\n            epochs_without_improvement = 0\n        else:\n            epochs_without_improvement += 1\n            \n        if epochs_without_improvement >= patience:\n            print(f'\\nEarly stopping at epoch {epoch+1} as there was no improvement in validation loss.')\n            break\n    \n    print('\\nTraining has completed!')\n    \n    return train_losses, valid_losses","metadata":{"execution":{"iopub.status.busy":"2023-06-18T14:19:49.828913Z","iopub.execute_input":"2023-06-18T14:19:49.829329Z","iopub.status.idle":"2023-06-18T14:19:49.840586Z","shell.execute_reply.started":"2023-06-18T14:19:49.829264Z","shell.execute_reply":"2023-06-18T14:19:49.839357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device","metadata":{"execution":{"iopub.status.busy":"2023-06-20T22:06:48.867087Z","iopub.execute_input":"2023-06-20T22:06:48.867675Z","iopub.status.idle":"2023-06-20T22:06:48.878551Z","shell.execute_reply.started":"2023-06-20T22:06:48.867619Z","shell.execute_reply":"2023-06-20T22:06:48.877527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#19a\nnb_epochs = 5\ntrain_losses, valid_losses = optimize(trainloaders,validloaders,model,criterion,optimizer,nb_epochs, class_weights)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-21T01:05:02.967963Z","iopub.execute_input":"2023-06-21T01:05:02.968611Z","iopub.status.idle":"2023-06-21T02:04:27.496996Z","shell.execute_reply.started":"2023-06-21T01:05:02.968568Z","shell.execute_reply":"2023-06-21T02:04:27.495809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(nb_epochs)\nprint(train_losses)\nprint(valid_losses)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:54:09.074311Z","iopub.execute_input":"2023-06-15T05:54:09.074992Z","iopub.status.idle":"2023-06-15T05:54:09.081055Z","shell.execute_reply.started":"2023-06-15T05:54:09.074942Z","shell.execute_reply":"2023-06-15T05:54:09.079934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot the graph of train_losses & valid_losses against nb_epochs. 20a\nepochs = range(1, nb_epochs+1)\nplt.plot(epochs, train_losses, 'g', label='Training loss')\nplt.plot(epochs, valid_losses, 'b', label='validation loss')\nplt.title('Training and Validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T00:45:06.164334Z","iopub.status.idle":"2023-06-19T00:45:06.165044Z","shell.execute_reply.started":"2023-06-19T00:45:06.164757Z","shell.execute_reply":"2023-06-19T00:45:06.164792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 21a\ndef test(dataloader,model):\n    '''\n    test function predicts the labels given an image batches.\n    \n    Args :\n         dataloader: DataLoader for the test_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         \n    Returns :\n         List of predicted labels.\n    '''\n    correct = 0\n    total = 0\n    model.eval() #Sets the model for evaluation.\n    \n    true_labels = []\n    predicted_labels = []\n    \n    with torch.no_grad():\n        \n        for batch,(x,y) in enumerate(dataloader):\n            #print(x)\n           # x = torch.tensor(x)\n            output = model(x.to(device))\n            \n            #predictions = output.argmax(dim=1).cpu().detach().tolist() #Predicted labels for an image batch.\n            predictions   = output.argmax(dim=1).cpu().detach() #Index for the highest score for all the samples in the batch.\n            total        += y.size(0)\n            correct      += (predictions == y.cpu().detach()).sum().item()\n            true_labels.extend(y.cpu().detach().tolist())\n            predicted_labels.extend(predictions.tolist())\n                \n    print('Testing has completed')\n    print(f'Accuracy on Training set = {100*(correct/total):.6f}% [{correct}/{total}]')        \n    return true_labels,predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:42:44.834439Z","iopub.execute_input":"2023-06-21T00:42:44.834947Z","iopub.status.idle":"2023-06-21T00:42:44.847639Z","shell.execute_reply.started":"2023-06-21T00:42:44.834901Z","shell.execute_reply":"2023-06-21T00:42:44.846387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#22a\ny_pred, pred = test(testloaders,model) #Calls the test function.","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:42:46.449953Z","iopub.execute_input":"2023-06-21T00:42:46.450911Z","iopub.status.idle":"2023-06-21T00:44:11.457858Z","shell.execute_reply.started":"2023-06-21T00:42:46.450853Z","shell.execute_reply":"2023-06-21T00:44:11.456672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#23a\nPATH = '/kaggle/working/inv3full2dataset.h5'\n# Save the model\n#torch.save(model.state_dict(), PATH)\ntorch.save(model.state_dict(), 'models/model_{}.bin'.format('model_name'))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T20:20:27.973957Z","iopub.execute_input":"2023-06-19T20:20:27.974574Z","iopub.status.idle":"2023-06-19T20:20:28.056414Z","shell.execute_reply.started":"2023-06-19T20:20:27.974514Z","shell.execute_reply":"2023-06-19T20:20:28.053816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#24a\nimport numpy as np\nfrom sklearn.metrics import f1_score, confusion_matrix\n\n# Ground truth labels\ntrue_labels = np.array(y_pred)\n\n# Predicted labels obtained from the test function\npredicted_labels = np.array(pred)\n\n# Calculate F1 score\nf1 = f1_score(true_labels, predicted_labels, average='micro')\n\n# Calculate confusion matrix\ncm = confusion_matrix(true_labels, predicted_labels)\n\nprint(\"F1 score:\", f1)\nprint(\"Confusion matrix:\")\nprint(cm)\nfrom sklearn.metrics import classification_report\nprint(classification_report(true_labels, predicted_labels))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T00:44:11.459835Z","iopub.execute_input":"2023-06-21T00:44:11.460856Z","iopub.status.idle":"2023-06-21T00:44:11.486208Z","shell.execute_reply.started":"2023-06-21T00:44:11.460814Z","shell.execute_reply":"2023-06-21T00:44:11.485013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(true_labels, predicted_labels))","metadata":{"execution":{"iopub.status.busy":"2023-06-16T00:22:54.651777Z","iopub.execute_input":"2023-06-16T00:22:54.652329Z","iopub.status.idle":"2023-06-16T00:22:54.667247Z","shell.execute_reply.started":"2023-06-16T00:22:54.652287Z","shell.execute_reply":"2023-06-16T00:22:54.665615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loss function\ncriterion = torch.nn.CrossEntropyLoss()\nbatch_size = 10\n# epochs\nmax_epochs = 15\nearly_stop = 5\n\n# learning rates\neta = 1e-3\n\n# scheduler\nstep  = 5\ngamma = 0.5","metadata":{"execution":{"iopub.status.busy":"2023-06-19T19:33:22.606675Z","iopub.execute_input":"2023-06-19T19:33:22.607396Z","iopub.status.idle":"2023-06-19T19:33:22.613445Z","shell.execute_reply.started":"2023-06-19T19:33:22.607355Z","shell.execute_reply":"2023-06-19T19:33:22.612186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#VALIDATION SETTINGS\n\n# placeholders\noof_preds = np.zeros((len(X_val), 5))\n\n# timer\ncv_start = time.time()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T19:33:19.149990Z","iopub.execute_input":"2023-06-19T19:33:19.150978Z","iopub.status.idle":"2023-06-19T19:33:19.156754Z","shell.execute_reply.started":"2023-06-19T19:33:19.150920Z","shell.execute_reply":"2023-06-19T19:33:19.155620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val","metadata":{"execution":{"iopub.status.busy":"2023-06-19T19:36:53.322384Z","iopub.execute_input":"2023-06-19T19:36:53.322857Z","iopub.status.idle":"2023-06-19T19:36:53.333483Z","shell.execute_reply.started":"2023-06-19T19:36:53.322812Z","shell.execute_reply":"2023-06-19T19:36:53.332250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#MODELING EPOCHS\n\n# placeholders\nval_kappas = []\nval_losses = []\ntrn_losses = []\nbad_epochs = 0\n\n# initialize and send to GPU\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = models.inception_v3(pretrained=True)\n#freezing feature layer\nfor param in model.parameters():\n    param.requires_grad = False\n    \n#modifying fc layer\nmodel.fc = torch.nn.Linear(in_features=2048, out_features=5, bias=True)\n#model.aux_logits = False\n#moving the model to device (cuda/cpu)\n#model.to(device)\nmodel = model.to(device)\n\n# optimizer\noptimizer = optim.Adam(model.parameters(), lr = eta)\nscheduler = lr_scheduler.StepLR(optimizer, step_size = step, gamma = gamma)\n\n# training and validation loop\nfor epoch in range(max_epochs):\n    ### PREPARATION\n\n    # timer\n    epoch_start = time.time()\n\n    # reset losses\n    trn_loss = 0.0\n    val_loss = 0.0\n\n    # placeholders\n    fold_preds = np.zeros((len(X_val), 5))\n\n\n    #TRAINING\n\n    # switch regime\n    model.train()\n\n    # loop through batches\n    for batch_i, data in enumerate(trainloaders):\n\n        # extract inputs and labels\n        inputs = data['image']\n        labels = data['label'].view(-1)\n        inputs = inputs.to(device, dtype = torch.float)\n        labels = labels.to(device, dtype = torch.long)\n        optimizer.zero_grad()\n\n        # forward and backward pass\n        with torch.set_grad_enabled(True):\n            outputs = model(inputs)\n            preds = outputs.logits\n            #preds = model(inputs)\n            #print(type(preds))\n            \n            loss  = criterion(preds, labels)\n            loss.backward()\n            optimizer.step()\n\n        # compute loss\n        trn_loss += loss.item() * inputs.size(0)\n        \n        \n    #INFERENCE\n\n    # switch regime\n    model.eval()\n    \n    # loop through batches\n    for batch_i, data in enumerate(validloaders):\n        \n        # extract inputs and labels\n        inputs = data['image']\n        labels = data['label'].view(-1)\n        inputs = inputs.to(device, dtype = torch.float)\n        labels = labels.to(device, dtype = torch.long)\n\n        # compute predictions\n        with torch.set_grad_enabled(False):\n            preds = model(inputs).detach()\n            fold_preds[batch_i * batch_size:(batch_i + 1) * batch_size, :] = preds.cpu().numpy()\n\n        # compute loss\n        loss      = criterion(preds, labels)\n        val_loss += loss.item() * inputs.size(0)\n        \n    # save predictions\n    oof_preds = fold_preds\n\n    # scheduler step\n    scheduler.step()\n\n\n    #EVALUATION\n\n    # evaluate performance\n    fold_preds_round = fold_preds.argmax(axis = 1)\n    val_kappa = metrics.cohen_kappa_score(y_val, fold_preds_round.astype('int'), weights = 'quadratic')\n\n    # save perfoirmance values\n    val_kappas.append(val_kappa)\n    val_losses.append(val_loss / len(X_val))\n    trn_losses.append(trn_loss / len(X_train))\n\n\n    #EARLY STOPPING\n\n    # display info\n    print('- epoch {}/{} | lr = {} | trn_loss = {:.4f} | val_loss = {:.4f} | val_kappa = {:.4f} | {:.2f} min'.format(\n        epoch + 1, max_epochs, scheduler.get_lr()[len(scheduler.get_lr()) - 1],\n        trn_loss / len(X_train), val_loss / len(X_val), val_kappa,\n        (time.time() - epoch_start) / 60))\n\n    # check if there is any improvement\n    if epoch > 0:       \n        if val_kappas[epoch] < val_kappas[epoch - bad_epochs - 1]:\n            bad_epochs += 1\n        else:\n            bad_epochs = 0\n\n    # save model weights if improvement\n    if bad_epochs == 0:\n        oof_preds_best = oof_preds.copy()\n        torch.save(model.state_dict(),'/kaggle/working/inv3full2dataset.h5')\n\n    # break if early stop\n    if bad_epochs == early_stop:\n        print('Early stopping. Best results: loss = {:.4f}, kappa = {:.4f} (epoch {})'.format(\n            np.min(val_losses), val_kappas[np.argmin(val_losses)], np.argmin(val_losses) + 1))\n        print('')\n        break\n\n    # break if max epochs\n    if epoch == (max_epochs - 1):\n        print('Did not met early stopping. Best results: loss = {:.4f}, kappa = {:.4f} (epoch {})'.format(\n            np.min(val_losses), val_kappas[np.argmin(val_losses)], np.argmin(val_losses) + 1))\n        print('')\n        break\n\n\n# load best predictions\noof_preds = oof_preds_best\n\n# print performance\nprint('')\nprint('Finished in {:.2f} minutes'.format((time.time() - cv_start) / 60))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T20:20:48.667270Z","iopub.execute_input":"2023-06-19T20:20:48.667656Z","iopub.status.idle":"2023-06-19T23:22:47.690723Z","shell.execute_reply.started":"2023-06-19T20:20:48.667621Z","shell.execute_reply":"2023-06-19T23:22:47.689644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#RECHECK PERFORMANCE\n\n# rounding\noof_preds_round = oof_preds.argmax(axis = 1)\ncoef = [0.5, 1.5, 2.5, 3.5]\nfor i, pred in enumerate(oof_preds_round):\n    if pred < coef[0]:\n        oof_preds_round[i] = 0\n    elif pred >= coef[0] and pred < coef[1]:\n        oof_preds_round[i] = 1\n    elif pred >= coef[1] and pred < coef[2]:\n        oof_preds_round[i] = 2\n    elif pred >= coef[2] and pred < coef[3]:\n        oof_preds_round[i] = 3\n    else:\n        oof_preds_round[i] = 4\n\n# compute kappa\ny_val_array = np.array(y_val.to_frame()[1])\noof_loss  = criterion(torch.tensor(oof_preds), torch.tensor(y_val_array).view(-1).type(torch.long))\noof_kappa = metrics.cohen_kappa_score(y_val_array, oof_preds_round.astype('int'), weights = 'quadratic')\nprint('OOF loss  = {:.4f}'.format(oof_loss))\nprint('OOF kappa = {:.4f}'.format(oof_kappa))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:48:42.966571Z","iopub.execute_input":"2023-06-19T23:48:42.967002Z","iopub.status.idle":"2023-06-19T23:48:43.013125Z","shell.execute_reply.started":"2023-06-19T23:48:42.966966Z","shell.execute_reply":"2023-06-19T23:48:43.011974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_preds_round = oof_preds.argmax(axis = 1)\ncoef = [0.5, 1.5, 2.5, 3.5]\nfor i, pred in enumerate(oof_preds_round):\n    if pred < coef[0]:\n        oof_preds_round[i] = 0\n    elif pred >= coef[0] and pred < coef[1]:\n        oof_preds_round[i] = 1\n    elif pred >= coef[1] and pred < coef[2]:\n        oof_preds_round[i] = 2\n    elif pred >= coef[2] and pred < coef[3]:\n        oof_preds_round[i] = 3\n    else:\n        oof_preds_round[i] = 4","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:57:14.624223Z","iopub.execute_input":"2023-06-19T23:57:14.624689Z","iopub.status.idle":"2023-06-19T23:57:14.658019Z","shell.execute_reply.started":"2023-06-19T23:57:14.624645Z","shell.execute_reply":"2023-06-19T23:57:14.656971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(oof_preds_round)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:57:31.658279Z","iopub.execute_input":"2023-06-19T23:57:31.658761Z","iopub.status.idle":"2023-06-19T23:57:31.669352Z","shell.execute_reply.started":"2023-06-19T23:57:31.658716Z","shell.execute_reply":"2023-06-19T23:57:31.667743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred, pred = test(testloaders,model) #Calls the test function.","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:50:20.342693Z","iopub.execute_input":"2023-06-19T23:50:20.343090Z","iopub.status.idle":"2023-06-19T23:51:33.644546Z","shell.execute_reply.started":"2023-06-19T23:50:20.343053Z","shell.execute_reply":"2023-06-19T23:51:33.643374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 21a\ndef test(dataloader,model):\n    '''\n    test function predicts the labels given an image batches.\n    \n    Args :\n         dataloader: DataLoader for the test_set.\n         model: Given an input produces an output by multiplying the input with the model weights.\n         \n    Returns :\n         List of predicted labels.\n    '''\n    correct = 0\n    total = 0\n    model.eval() #Sets the model for evaluation.\n    \n    true_labels = []\n    predicted_labels = []\n    \n    with torch.no_grad():\n        \n        for batch_i, data in enumerate(validloaders):\n        \n        # extract inputs and labels\n            inputs = data['image']\n            labels = data['label'].view(-1)\n            inputs = inputs.to(device, dtype = torch.float)\n            labels = labels.to(device, dtype = torch.long)\n            output = model(inputs)\n            #preds = outputs.logits\n            #preds = model(inputs)\n            #print(type(preds))\n            \n            #loss  = criterion(preds, labels)\n            \n            #output = model(x.to(device))\n            \n            #predictions = output.argmax(dim=1).cpu().detach().tolist() #Predicted labels for an image batch.\n            preds   = output.argmax(dim=1).cpu().detach() #Index for the highest score for all the samples in the batch.\n            total        += preds.size(0)\n            correct      += (preds == preds.cpu().detach()).sum().item()\n            true_labels.extend(preds.cpu().detach().tolist())\n            predicted_labels.extend(preds.tolist())\n                \n    print('Testing has completed')\n    print(f'Accuracy on Training set = {100*(correct/total):.6f}% [{correct}/{total}]')        \n    return true_labels,predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:50:18.530488Z","iopub.execute_input":"2023-06-19T23:50:18.531448Z","iopub.status.idle":"2023-06-19T23:50:18.544325Z","shell.execute_reply.started":"2023-06-19T23:50:18.531390Z","shell.execute_reply":"2023-06-19T23:50:18.542982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#24a\nimport numpy as np\nfrom sklearn.metrics import f1_score, confusion_matrix\n\n# Ground truth labels\ntrue_labels = np.array(y_pred)\n\n# Predicted labels obtained from the test function\npredicted_labels = np.array(pred)\n\n# Calculate F1 score\nf1 = f1_score(true_labels, predicted_labels, average='micro')\n\n# Calculate confusion matrix\ncm = confusion_matrix(true_labels, predicted_labels)\n\nprint(\"F1 score:\", f1)\nprint(\"Confusion matrix:\")\nprint(cm)\nfrom sklearn.metrics import classification_report\nprint(classification_report(true_labels, predicted_labels))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T23:51:33.646708Z","iopub.execute_input":"2023-06-19T23:51:33.647963Z","iopub.status.idle":"2023-06-19T23:51:33.683451Z","shell.execute_reply.started":"2023-06-19T23:51:33.647919Z","shell.execute_reply":"2023-06-19T23:51:33.682180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install -U scikit-learn scipy matplotlib","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:32:54.615742Z","iopub.execute_input":"2023-06-21T08:32:54.616742Z","iopub.status.idle":"2023-06-21T08:33:08.994536Z","shell.execute_reply.started":"2023-06-21T08:32:54.616703Z","shell.execute_reply":"2023-06-21T08:33:08.993480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip uninstall openslide-python\ny\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:35:39.404054Z","iopub.execute_input":"2023-06-21T08:35:39.404625Z","iopub.status.idle":"2023-06-21T08:35:39.410003Z","shell.execute_reply.started":"2023-06-21T08:35:39.404590Z","shell.execute_reply":"2023-06-21T08:35:39.408979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\n#import openslide\nimport os\nimport tensorflow as tf\n\n\n\nfrom random import randint\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, confusion_matrix\nfrom tqdm import tqdm\n%matplotlib inline\n\nprint(tf.__version__)\nprint(tf.keras.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:36:02.261898Z","iopub.execute_input":"2023-06-21T08:36:02.262643Z","iopub.status.idle":"2023-06-21T08:36:02.273638Z","shell.execute_reply.started":"2023-06-21T08:36:02.262606Z","shell.execute_reply":"2023-06-21T08:36:02.272773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\n\n\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('diabetic-retinopathy-resized')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:36:05.149658Z","iopub.execute_input":"2023-06-21T08:36:05.150052Z","iopub.status.idle":"2023-06-21T08:36:05.445403Z","shell.execute_reply.started":"2023-06-21T08:36:05.150018Z","shell.execute_reply":"2023-06-21T08:36:05.444614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tpu = tf.distribute.cluster_resolver.TPUClusterResolver() ","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:34:06.578664Z","iopub.execute_input":"2023-06-21T08:34:06.579050Z","iopub.status.idle":"2023-06-21T08:34:06.860541Z","shell.execute_reply.started":"2023-06-21T08:34:06.579017Z","shell.execute_reply":"2023-06-21T08:34:06.859346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/diabetic-retinopathy-resized/trainLabels.csv')\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:30:43.165925Z","iopub.status.idle":"2023-06-21T08:30:43.166294Z","shell.execute_reply.started":"2023-06-21T08:30:43.166095Z","shell.execute_reply":"2023-06-21T08:30:43.166113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain,test = train_test_split(train_df,test_size = 0.16,random_state=1,stratify = train_df['level'])","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:30:38.130705Z","iopub.execute_input":"2023-06-21T07:30:38.134304Z","iopub.status.idle":"2023-06-21T07:30:38.186030Z","shell.execute_reply.started":"2023-06-21T07:30:38.134238Z","shell.execute_reply":"2023-06-21T07:30:38.184822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape) \nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:30:48.786998Z","iopub.execute_input":"2023-06-21T07:30:48.787415Z","iopub.status.idle":"2023-06-21T07:30:48.793995Z","shell.execute_reply.started":"2023-06-21T07:30:48.787374Z","shell.execute_reply":"2023-06-21T07:30:48.792605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:30:55.030717Z","iopub.execute_input":"2023-06-21T07:30:55.031258Z","iopub.status.idle":"2023-06-21T07:30:55.060328Z","shell.execute_reply.started":"2023-06-21T07:30:55.031204Z","shell.execute_reply":"2023-06-21T07:30:55.059310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train,valid = train_test_split(train,test_size = 0.168,random_state=1,stratify = train['level'])\ntrain_paths = train[\"image\"].apply(lambda x: '/kaggle/input/diabetic-retinopathy-resized' + '/resized_train_cropped/resized_train_cropped/' + x + '.jpeg').values\nvalid_paths = valid[\"image\"].apply(lambda x: '/kaggle/input/diabetic-retinopathy-resized' + '/resized_train/resized_train/' + x + '.jpeg').values","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:34:19.556565Z","iopub.execute_input":"2023-06-21T07:34:19.557182Z","iopub.status.idle":"2023-06-21T07:34:19.595795Z","shell.execute_reply.started":"2023-06-21T07:34:19.557142Z","shell.execute_reply":"2023-06-21T07:34:19.594592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:34:23.731476Z","iopub.execute_input":"2023-06-21T07:34:23.732086Z","iopub.status.idle":"2023-06-21T07:34:23.741407Z","shell.execute_reply.started":"2023-06-21T07:34:23.732038Z","shell.execute_reply":"2023-06-21T07:34:23.740055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.get_dummies(train['level']).astype('int32').values\nvalid_labels = pd.get_dummies(valid['level']).astype('int32').values\n\nprint(train_labels.shape) \nprint(valid_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:34:39.105797Z","iopub.execute_input":"2023-06-21T07:34:39.107004Z","iopub.status.idle":"2023-06-21T07:34:39.124733Z","shell.execute_reply.started":"2023-06-21T07:34:39.106951Z","shell.execute_reply":"2023-06-21T07:34:39.123498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:34:43.338172Z","iopub.execute_input":"2023-06-21T07:34:43.338677Z","iopub.status.idle":"2023-06-21T07:34:43.349632Z","shell.execute_reply.started":"2023-06-21T07:34:43.338630Z","shell.execute_reply":"2023-06-21T07:34:43.348198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE= 16\nimg_size = 512\nEPOCHS = 50\nnb_classes = 5","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:44:39.966439Z","iopub.execute_input":"2023-06-21T07:44:39.967682Z","iopub.status.idle":"2023-06-21T07:44:39.974169Z","shell.execute_reply.started":"2023-06-21T07:44:39.967616Z","shell.execute_reply":"2023-06-21T07:44:39.972953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport random\n\n# IMAGE PREPROCESSING\ndef prepare_image(path, sigmaX=10, do_random_crop=False):\n    \"\"\"\n    Preprocess image\n    \"\"\"\n    # Read image\n    image = tf.io.read_file(path)\n    image = tf.image.decode_jpeg(image, channels=3)\n    \n    # Perform smart crops\n    image = crop_black(image, tol=7)\n    if do_random_crop:\n        image = random_crop(image, size=(0.9, 1))\n    \n    # Resize and color\n   # image = tf.image.adjust_brightness(image, 0.2)\n  \n    \n   # image = tf.image.resize(image, size=(512, 512))\n    #image = tf.image.convert_image_dtype(image, tf.float32)\n    #image = tf.image.adjust_contrast(image, contrast_factor=2)\n    #image = tf.image.random_flip_left_right(image)\n    #image = tf.image.random_flip_up_down(image)\n    #image = tf.cast(image, tf.float32) / 255.0\n    \n    image = tf.image.adjust_brightness(image, 0.2)\n    image = tf.image.resize(image, size=(512, 512))\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    image = tf.image.adjust_contrast(image, contrast_factor=2)\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_hue(image, max_delta=0.08)\n    image = tf.image.random_saturation(image, lower=0.6, upper=1.4)\n    image = tf.image.random_contrast(image, lower=0.6, upper=1.4)\n    image = tf.image.random_brightness(image, max_delta=0.08)\n    image = tf.cast(image, tf.float32) / 255.0\n    # Circular crop\n    #image = circle_crop(image, sigmaX=sigmaX)\n    \n    return image\n\n# CROP FUNCTIONS\ndef crop_black(img, tol=7):\n    \"\"\"\n    Perform automatic crop of black areas\n    \"\"\"\n    mask = tf.reduce_max(img, axis=2) > tol\n    mask = tf.expand_dims(mask, axis=-1)\n    img = tf.cast(img, tf.float32)\n    img = img * tf.cast(mask, tf.float32)\n    return img\n\ndef circle_crop(img, sigmaX=10):\n    \"\"\"\n    Perform circular crop around image center\n    \"\"\"\n    height, width, _ = img.shape\n    largest_side = tf.reduce_max([height, width])\n    img = tf.image.resize(img, size=(largest_side, largest_side))\n    \n    x = width // 2\n    y = height // 2\n    r = tf.minimum(x, y)\n    \n    mask = tf.numpy_function(create_circular_mask, [height, width, x, y, r], tf.bool)\n    img = tf.cast(img, tf.float32)\n    img = img * tf.cast(mask, tf.float32)\n    return img\n\ndef create_circular_mask(height, width, center=None, radius=None):\n    if center is None:\n        center = (width // 2, height // 2)\n    if radius is None:\n        radius = min(center[0], center[1], width - center[0], height - center[1])\n\n    Y, X = tf.meshgrid(tf.range(height), tf.range(width))\n    dist_from_center = tf.sqrt((X - center[0])**2 + (Y - center[1])**2)\n    mask = tf.cast(dist_from_center <= radius, tf.float32)\n\n    return mask\n\ndef random_crop(img, size=(0.9, 1)):\n    \"\"\"\n    Random crop\n    \"\"\"\n    height, width, _ = img.shape\n    \n    cut = 1 - random.uniform(size[0], size[1])\n    i = tf.random.uniform(shape=[], maxval=int(cut * height), dtype=tf.int32)\n    j = tf.random.uniform(shape=[], maxval=int(cut * width), dtype=tf.int32)\n    h = i + int((1 - cut) * height)\n    w = j + int((1 - cut) * width)\n    \n    img = img[i:h, j:w, :]\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:17:17.022783Z","iopub.execute_input":"2023-06-21T08:17:17.023175Z","iopub.status.idle":"2023-06-21T08:17:17.047293Z","shell.execute_reply.started":"2023-06-21T08:17:17.023137Z","shell.execute_reply":"2023-06-21T08:17:17.046053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(filename, label=None, image_size=(512, 512)):\n    #bits = tf.io.read_file(filename)\n    #image = tf.image.decode_jpeg(bits, channels=3)\n    \n    image = prepare_image(filename)\n    #image = tf.cast(image, tf.float32) / 255.0\n    #image = tf.image.resize(image, (512, 512))\n    if label is None:\n        return image\n    else:\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:17:22.417588Z","iopub.execute_input":"2023-06-21T08:17:22.418338Z","iopub.status.idle":"2023-06-21T08:17:22.425665Z","shell.execute_reply.started":"2023-06-21T08:17:22.418290Z","shell.execute_reply":"2023-06-21T08:17:22.424071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:44:51.065041Z","iopub.execute_input":"2023-06-21T07:44:51.066001Z","iopub.status.idle":"2023-06-21T07:44:51.071605Z","shell.execute_reply.started":"2023-06-21T07:44:51.065943Z","shell.execute_reply":"2023-06-21T07:44:51.070186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((train_paths, train_labels))\n    .map(decode_image, num_parallel_calls=AUTO)\n    .repeat()\n    .cache()\n    .shuffle(512)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n    )","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:17:23.826666Z","iopub.execute_input":"2023-06-21T08:17:23.827066Z","iopub.status.idle":"2023-06-21T08:17:24.169914Z","shell.execute_reply.started":"2023-06-21T08:17:23.827029Z","shell.execute_reply":"2023-06-21T08:17:24.168736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset","metadata":{"execution":{"iopub.status.busy":"2023-06-21T07:46:55.426797Z","iopub.execute_input":"2023-06-21T07:46:55.427283Z","iopub.status.idle":"2023-06-21T07:46:55.441430Z","shell.execute_reply.started":"2023-06-21T07:46:55.427235Z","shell.execute_reply":"2023-06-21T07:46:55.439967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Iterate over the dataset and display images\nfor images, labels in train_dataset.take(1):  # Take the first batch\n    for image in images:\n        plt.figure()\n        plt.imshow(image.numpy())\n        plt.axis('off')\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:17:25.572712Z","iopub.execute_input":"2023-06-21T08:17:25.574011Z","iopub.status.idle":"2023-06-21T08:17:41.935448Z","shell.execute_reply.started":"2023-06-21T08:17:25.573961Z","shell.execute_reply":"2023-06-21T08:17:41.934412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((valid_paths, valid_labels))\n    .map(decode_image, num_parallel_calls=AUTO)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:18:00.794629Z","iopub.execute_input":"2023-06-21T08:18:00.795031Z","iopub.status.idle":"2023-06-21T08:18:00.880593Z","shell.execute_reply.started":"2023-06-21T08:18:00.794992Z","shell.execute_reply":"2023-06-21T08:18:00.879577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strategy = tf.distribute.MirroredStrategy()\n    with strategy.scope():\n        resnet = ResNet50(include_top = False, weights = 'imagenet', input_tensor = None, input_shape = (512,512,3))\n        x = resnet.output\n        x = Flatten()(x)\n        output_layer = Dense(5,activation = 'softmax',name = 'softmax')(x)\n        final_model = keras.Model(inputs = resnet.input, outputs = output_layer)\n        opt = keras.optimizers.Nadam(lr = 0.0001,beta_1=0.9,beta_2=0.9)\n        final_model.compile(loss='categorical_crossentropy', optimizer = opt, metrics=['accuracy'])\n        return final_model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.optimizers import *\nimport keras\nfrom keras.layers.convolutional import Conv2D, MaxPooling2D\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Activation, Flatten\nfrom tensorflow.keras.applications import ResNet50\nstrategy = tf.distribute.MirroredStrategy()\ndef get_resnet():\n    \n    \n    with strategy.scope():\n        resnet = ResNet50(include_top = False, weights = 'imagenet', input_tensor = None, input_shape = (512,512,3))\n        x = resnet.output\n        x = Flatten()(x)\n        output_layer = Dense(5,activation = 'softmax',name = 'softmax')(x)\n        final_model = keras.Model(inputs = resnet.input, outputs = output_layer)\n        opt = keras.optimizers.Nadam(lr = 0.0001,beta_1=0.9,beta_2=0.9)\n        final_model.compile(loss='categorical_crossentropy', optimizer = opt, metrics=['accuracy'])\n        return final_model\n\ndef get_xception():\n    with strategy.scope():\n        resnet = keras.applications.Xception(include_top = False, weights = 'imagenet', input_tensor = None, input_shape = (512,512,3))\n        x = resnet.output\n        x = Flatten()(x)\n        output_layer = Dense(5,activation = 'softmax',name = 'softmax')(x)\n        final_model = keras.Model(inputs = resnet.input, outputs = output_layer)\n        opt = keras.optimizers.Nadam(lr = 0.0001,beta_1=0.9,beta_2=0.9)\n        final_model.compile(loss='categorical_crossentropy', optimizer = opt, metrics=['accuracy'])\n        return final_model\n    \ndef get_inception():\n    with strategy.scope():\n        resnet = keras.applications.InceptionV3(include_top = False, weights = 'imagenet', input_tensor = None, input_shape = (512,512,3))\n        x = resnet.output\n        x = Flatten()(x)\n        output_layer = Dense(5,activation = 'softmax',name = 'softmax')(x)\n        final_model = keras.Model(inputs = resnet.input, outputs = output_layer)\n        opt = keras.optimizers.Nadam(lr = 0.0001,beta_1=0.9,beta_2=0.9)\n        final_model.compile(loss='categorical_crossentropy', optimizer = opt, metrics=['accuracy'])\n        return final_model\n    \ndef get_dense121():\n    with strategy.scope():\n        resnet = keras.applications.DenseNet121(include_top = False, weights = 'imagenet', input_tensor = None, input_shape = (512,512,3))\n        x = resnet.output\n        x = Flatten()(x)\n        output_layer = Dense(5,activation = 'softmax',name = 'softmax')(x)\n        final_model = keras.Model(inputs = resnet.input, outputs = output_layer)\n        opt = keras.optimizers.Nadam(lr = 0.0001,beta_1=0.9,beta_2=0.9)\n        final_model.compile(loss='categorical_crossentropy', optimizer = opt, metrics=['accuracy'])\n        return final_model\n    \ndef get_dense169():\n    with strategy.scope():\n        resnet = keras.applications.DenseNet169(include_top = False, weights = 'imagenet', input_tensor = None, input_shape = (512,512,3))\n        x = resnet.output\n        x = Flatten()(x)\n        output_layer = Dense(5,activation = 'softmax',name = 'softmax')(x)\n        final_model = keras.Model(inputs = resnet.input, outputs = output_layer)\n        opt = keras.optimizers.Nadam(lr = 0.0001,beta_1=0.9,beta_2=0.9)\n        final_model.compile(loss='categorical_crossentropy', optimizer = opt, metrics=['accuracy'])\n        return final_model\n    \nresnetmodel = get_resnet()\n#xceptionmodel = get_xception()\n#inceptionmodel = get_inception()\n#dense121model = get_dense121()\n#dense169model = get_dense169()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:21:49.222967Z","iopub.execute_input":"2023-06-21T08:21:49.223442Z","iopub.status.idle":"2023-06-21T08:21:53.502622Z","shell.execute_reply.started":"2023-06-21T08:21:49.223396Z","shell.execute_reply":"2023-06-21T08:21:53.501535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import *\n\nes = EarlyStopping(monitor = 'val_loss',verbose = 1,mode='min')\nplat = ReduceLROnPlateau(monitor = 'val_loss',verbose=1,factor=0.1,min_lr = 0.00001,patience=5)\nCheckpoint= ModelCheckpoint(\"./resnetmodel_adam_0001.h5\", monitor='val_loss', verbose=1, save_best_only=True,\n       save_weights_only=True,mode='min')\n\n\ntrain_history1 = resnetmodel.fit(\n            train_dataset, \n            validation_data = valid_dataset, \n            steps_per_epoch=train_labels.shape[0] // BATCH_SIZE,            \n            validation_steps=valid_labels.shape[0] // BATCH_SIZE,            \n            callbacks=[es,plat,Checkpoint],\n            epochs=EPOCHS,\n            verbose=1\n            )","metadata":{"execution":{"iopub.status.busy":"2023-06-21T08:22:09.655783Z","iopub.execute_input":"2023-06-21T08:22:09.656486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}