{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# for system settings\nimport os\nimport warnings\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore')\nimport time\n# for excessing or creating tabular data\nimport pandas as pd\nimport pandas.util.testing as tm\n# for matrix manipulation\nimport numpy as np\n# for visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n# for splitting the data\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:12:29.306684Z","iopub.execute_input":"2023-05-21T15:12:29.307032Z","iopub.status.idle":"2023-05-21T15:12:29.313597Z","shell.execute_reply.started":"2023-05-21T15:12:29.307005Z","shell.execute_reply":"2023-05-21T15:12:29.311594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/aptos2019-blindness-detection/train.csv\")\nprint(\"train.csv:\")\nprint(\"Number of Training images: {}\\n\".format(train_data.shape[0]))\nprint(train_data.head(2),\"\\n\")\nprint(\"-\"*100)\ntest_data = pd.read_csv(\"/kaggle/input/aptos2019-blindness-detection/test.csv\")\nprint(\"test.csv: \")\nprint(\"Number of Testing images: {}\\n\".format(test_data.shape[0]))\nprint(test_data.head(2))\nprint(\"\\n\",\"-\"*100)\nprint(\"sample_submission.csv:\")\nsample_submission = pd.read_csv(\"/kaggle/input/aptos2019-blindness-detection/sample_submission.csv\")\nprint(\"The format of submitting the final predictions on testing images: \")\nprint(sample_submission.head(5))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:13:47.286554Z","iopub.execute_input":"2023-05-21T15:13:47.287376Z","iopub.status.idle":"2023-05-21T15:13:47.340288Z","shell.execute_reply.started":"2023-05-21T15:13:47.287327Z","shell.execute_reply":"2023-05-21T15:13:47.339317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Intialization of variables which are useful for the later tasks.\nimg_width = 512\nimg_height = 512\nno_channels = 3\nsplit_size = 0.15\nclass_labels = {0: 'No DR[0]',1: 'Mild[1]', 2: 'Moderate[2]', 3: 'Severe[3]', 4: 'Proliferative DR[4]'}","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:14:16.276830Z","iopub.execute_input":"2023-05-21T15:14:16.277198Z","iopub.status.idle":"2023-05-21T15:14:16.282227Z","shell.execute_reply.started":"2023-05-21T15:14:16.277167Z","shell.execute_reply":"2023-05-21T15:14:16.281332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = [len(train_data), len(test_data)]\nprint(\"Number of Images in train dataset: \", data[0])\nprint(\"Number of Images in test dataset: \", data[1])\nlabels = ['train_data','test_data']\nplt.pie(data,explode = [0,0.1], labels= labels, shadow = True, colors = ['yellowgreen','gold'],autopct='%1.1f%%', startangle = 120)\nplt.title('Pie Chart Analysis of size of train and test datasets')\nplt.axis('equal')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:14:21.821171Z","iopub.execute_input":"2023-05-21T15:14:21.821877Z","iopub.status.idle":"2023-05-21T15:14:22.021179Z","shell.execute_reply.started":"2023-05-21T15:14:21.821841Z","shell.execute_reply":"2023-05-21T15:14:22.020273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels_ = list(set(train_data['diagnosis'])) \nprint(\"Number of target classes: {}\".format(class_labels_))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:14:37.830111Z","iopub.execute_input":"2023-05-21T15:14:37.830494Z","iopub.status.idle":"2023-05-21T15:14:37.839310Z","shell.execute_reply.started":"2023-05-21T15:14:37.830463Z","shell.execute_reply":"2023-05-21T15:14:37.838405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_sizes = []\nfor i in range(0,5):\n    class_sizes.append(list(train_data['diagnosis']).count(i))\nlabels = class_labels.values()\ncolors = ['gold', 'yellowgreen', 'lightcoral', 'lightskyblue','darkgreen']\nplt.pie(class_sizes,explode = [0.1,0,0,0,0], labels= labels, shadow = True,autopct='%1.1f%%', startangle = 35)\nplt.title('Pie Chart Analysis of Number of Images on each target label:')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:14:47.521029Z","iopub.execute_input":"2023-05-21T15:14:47.521406Z","iopub.status.idle":"2023-05-21T15:14:47.740901Z","shell.execute_reply.started":"2023-05-21T15:14:47.521355Z","shell.execute_reply":"2023-05-21T15:14:47.739547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_analysis(dataframe, path):\n    width_range = []\n    height_range = []\n    for i in range(dataframe.shape[0]):\n        img = cv2.imread(path+dataframe.iloc[i]['id_code']+'.png')\n        height, width, _ = img.shape\n        width_range.append(width)\n        height_range.append(height)\n    return width_range, height_range\nwidth_range, height_range = image_analysis(train_data, '/kaggle/input/aptos2019-blindness-detection/train_images/')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:16:38.322346Z","iopub.execute_input":"2023-05-21T15:16:38.322763Z","iopub.status.idle":"2023-05-21T15:23:27.628059Z","shell.execute_reply.started":"2023-05-21T15:16:38.322730Z","shell.execute_reply":"2023-05-21T15:23:27.627061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_width = sum(width_range)/len(width_range)\navg_height = sum(height_range)/len(height_range)\nmax_width = max(width_range)\nmax_height = max(height_range)\nmin_width = min(width_range)\nmin_height = min(height_range)\nprint(\"Average width of images in training set: {}\".format(int(avg_width)))\nprint(\"Average height of images in training set: {}\".format(int(avg_height)))\nprint(\"-\"*100)\nprint(\"Maximum width of images in training set: {}\".format(max_width))\nprint(\"Maximum height of images in training set: {}\".format(max_height))\nprint(\"-\"*100)\nprint(\"Minimum width of images in training set: {}\".format(min_width))\nprint(\"Minimum height of images in training set: {}\".format(min_height))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:24:00.434213Z","iopub.execute_input":"2023-05-21T15:24:00.435196Z","iopub.status.idle":"2023-05-21T15:24:00.444108Z","shell.execute_reply.started":"2023-05-21T15:24:00.435161Z","shell.execute_reply":"2023-05-21T15:24:00.443126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,5))\nplt.subplot(1,2,1)\nsns.distplot(width_range, kde = False, label = 'train_width')\nsns.distplot(height_range, kde = False, label = 'train_height')\nplt.legend()\nplt.title(\"Histogram of Height and Width in Training Images\")\nplt.subplot(1,2,2)\nsns.kdeplot(width_range, label = 'train_width')\nsns.kdeplot(height_range, label = 'train_height')\nplt.legend()\nplt.title('KDE plot of Height and Width in Training Images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:24:21.975011Z","iopub.execute_input":"2023-05-21T15:24:21.975395Z","iopub.status.idle":"2023-05-21T15:24:22.591594Z","shell.execute_reply.started":"2023-05-21T15:24:21.975342Z","shell.execute_reply":"2023-05-21T15:24:22.590697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"width_range_test, height_range_test = image_analysis(test_data, '/kaggle/input/aptos2019-blindness-detection/test_images/')\navg_width = sum(width_range_test)/len(width_range_test)\navg_height = sum(height_range_test)/len(height_range_test)\nmax_width = max(width_range_test)\nmax_height = max(height_range_test)\nmin_width = min(width_range_test)\nmin_height = min(height_range_test)\nprint(\"Average width of images in training set: {}\".format(int(avg_width)))\nprint(\"Average height of images in training set: {}\".format(int(avg_height)))\nprint('-'*100)\nprint(\"Maximum width of images in test set: {}\".format(max_width))\nprint(\"Maximum height of images in test set: {}\".format(max_height))\nprint('-'*100)\nprint(\"Minimum width of images in test set: {}\".format(min_width))\nprint(\"Minimum height of images in test set: {}\".format(min_height))","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:24:57.672736Z","iopub.execute_input":"2023-05-21T15:24:57.673085Z","iopub.status.idle":"2023-05-21T15:26:36.128076Z","shell.execute_reply.started":"2023-05-21T15:24:57.673057Z","shell.execute_reply":"2023-05-21T15:26:36.127051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,5))\nplt.subplot(1,2,1)\nsns.distplot(width_range_test, kde = False, label = 'test_width')\nsns.distplot(height_range_test, kde = False, label = 'test_height')\nplt.legend()\nplt.title(\"Histogram of Height and Width in Test Images\")\nplt.subplot(1,2,2)\nsns.kdeplot(width_range_test, label = 'test_width')\nsns.kdeplot(height_range_test, label = 'test_height')\nplt.legend()\nplt.title('KDE plot of Height and Width in Test Images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:27:50.389305Z","iopub.execute_input":"2023-05-21T15:27:50.390066Z","iopub.status.idle":"2023-05-21T15:27:51.208107Z","shell.execute_reply.started":"2023-05-21T15:27:50.390030Z","shell.execute_reply":"2023-05-21T15:27:51.207266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def splitting_data(train_data, size, is_split = True):\n    \"\"\"\n       This function splits the given data into train and validation sets basing on size for validation.\n       Args : df - (dataframe) through which splitting is performed \n            size - (Integer) test_size -> percentage of data for validation set \n            is_split = (boolean) returns train and validation if it is True , otherwise it simply returns the train data\n       Outputs : (Series Object) train and validation sets of data \n\n    \"\"\"\n    try:\n        if is_split:\n            data = train_data['id_code']\n            labels = train_data['diagnosis']\n            train_x, validation_x, train_labels, validation_labels = train_test_split(data, labels, stratify=labels, shuffle=True, test_size=size)\n            print(\"Training data: {} {}\".format(train_x.shape, train_labels.shape))\n            print(\"Validation data: {} {}\".format(validation_x.shape,validation_labels.shape))\n            return train_x, train_labels, validation_x, validation_labels\n        else:\n            return train_data['id_code'], train_data['diagnosis'], [], []\n    except:\n        print(\"Error: Invalid file format, Function argument requires .csv file!!!\")","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:28:21.669791Z","iopub.execute_input":"2023-05-21T15:28:21.670161Z","iopub.status.idle":"2023-05-21T15:28:21.680465Z","shell.execute_reply.started":"2023-05-21T15:28:21.670125Z","shell.execute_reply":"2023-05-21T15:28:21.679421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x, train_labels, validation_x, validation_labels = splitting_data(train_data, split_size)   # function calling","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:28:46.615556Z","iopub.execute_input":"2023-05-21T15:28:46.615908Z","iopub.status.idle":"2023-05-21T15:28:46.631861Z","shell.execute_reply.started":"2023-05-21T15:28:46.615879Z","shell.execute_reply":"2023-05-21T15:28:46.629718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pwd","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:29:45.236461Z","iopub.execute_input":"2023-05-21T15:29:45.236928Z","iopub.status.idle":"2023-05-21T15:29:46.235395Z","shell.execute_reply.started":"2023-05-21T15:29:45.236872Z","shell.execute_reply":"2023-05-21T15:29:46.234245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.DataFrame(train_x, columns = ['id_code'])\ntrain['diagnosis'] = train_labels\ntrain.to_csv(\"/kaggle/working/training.csv\", index = False)\nvalidation = pd.DataFrame(validation_x, columns = ['id_code'])\nvalidation['diagnosis'] = validation_labels\nvalidation.to_csv('/kaggle/working/validation.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:30:26.396259Z","iopub.execute_input":"2023-05-21T15:30:26.396669Z","iopub.status.idle":"2023-05-21T15:30:26.425023Z","shell.execute_reply.started":"2023-05-21T15:30:26.396635Z","shell.execute_reply":"2023-05-21T15:30:26.424165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')\ntest.to_csv('/kaggle/working/test.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:32:29.403129Z","iopub.execute_input":"2023-05-21T15:32:29.403514Z","iopub.status.idle":"2023-05-21T15:32:29.422233Z","shell.execute_reply.started":"2023-05-21T15:32:29.403485Z","shell.execute_reply":"2023-05-21T15:32:29.421420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/working/training.csv\")\nvalidation = pd.read_csv(\"/kaggle/working/validation.csv\")\ntrain_x = train['id_code']\ntrain_labels = train['diagnosis']\nvalidation_x = validation['id_code']\nvalidation_labels = validation['diagnosis']","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:33:17.363583Z","iopub.execute_input":"2023-05-21T15:33:17.364282Z","iopub.status.idle":"2023-05-21T15:33:17.378076Z","shell.execute_reply.started":"2023-05-21T15:33:17.364248Z","shell.execute_reply":"2023-05-21T15:33:17.377062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Target Labels Analysis on Train and validation sets:","metadata":{}},{"cell_type":"code","source":"def class_analysis(labels, d_set):\n    \"\"\"\n    This function plots the histogram of class labels for given set of labels.\n    Args : labels - (Series object) which contains the class_labels of train or validation sets.\n           d_set - (String) which helps to known whether it is a train or validation set.\n    Output : None - this function doesn't return anything\n    \"\"\"\n    if d_set == 'training': print(\"-\"*100,'\\n')\n    counter = labels.value_counts().sort_index()\n    counter.plot(kind = 'bar')\n    plt.title('Number of images for each Class label in {} set'.format(d_set))\n    plt.xlabel('Classes')\n    plt.ylabel('Number of images')\n    plt.grid()\n    plt.show()\n    iter=0\n\n    for i in list(set(labels)):\n        percentage = list(labels).count(i)/len(list(labels))\n        print(\"Number of images in class - {} ({}) , nearly {} % of total data\".format(i,class_labels[i],np.round(percentage*100,4)))\n        iter+=1\n    if d_set == 'training':\n        print(\"\\n\",\"=\"*100,\"\\n\")\n    if d_set == 'validation': print(\"-\"*100,'\\n')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:33:42.309211Z","iopub.execute_input":"2023-05-21T15:33:42.310126Z","iopub.status.idle":"2023-05-21T15:33:42.319972Z","shell.execute_reply.started":"2023-05-21T15:33:42.310077Z","shell.execute_reply":"2023-05-21T15:33:42.318953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_analysis(train_labels,'training')\nclass_analysis(validation_labels,'validation')","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:33:50.810202Z","iopub.execute_input":"2023-05-21T15:33:50.811050Z","iopub.status.idle":"2023-05-21T15:33:51.383779Z","shell.execute_reply.started":"2023-05-21T15:33:50.811015Z","shell.execute_reply":"2023-05-21T15:33:51.382819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image Preprocessing:","metadata":{}},{"cell_type":"code","source":"class ImageProcessing:\n    def __init__(self, img_height, img_width, no_channels, tol=7, sigmaX=8):\n\n        ''' Initialzation of variables'''\n\n        self.img_height = img_height\n        self.img_width = img_width\n        self.no_channels = no_channels\n        self.tol = tol\n        self.sigmaX = sigmaX\n\n    def cropping_2D(self, img, is_cropping = False):\n\n        '''This function is used for Cropping the extra dark part of the GRAY images'''\n\n        mask = img>self.tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n\n    def cropping_3D(self, img, is_cropping = False):\n\n        '''This function is used for Cropping the extra dark part of the RGB images'''\n\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>self.tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # if image is too dark we return the image\n            return img \n        else:\n            img1 = img[:,:,0][np.ix_(mask.any(1),mask.any(0))]  #for channel_1 (R)\n            img2 = img[:,:,1][np.ix_(mask.any(1),mask.any(0))]  #for channel_2 (G)\n            img3 = img[:,:,2][np.ix_(mask.any(1),mask.any(0))]  #for channel_3 (B)         \n            img = np.stack([img1,img2,img3],axis=-1)\n        return img\n\n    def Gaussian_blur(self, img, is_gaussianblur = False):\n\n        '''This function is used for adding Gaussian blur (image smoothing technique) which helps in reducing noise in the image.'''\n\n        img = cv2.addWeighted(img,4,cv2.GaussianBlur(img,(0,0),self.sigmaX),-4,128)\n        return img\n\n    def draw_circle(self,img, is_drawcircle = True):\n\n        '''This function is used for drawing a circle from the center of the image.'''\n\n        x = int(self.img_width/2)\n        y = int(self.img_height/2)\n        r = np.amin((x,y))     # finding radius to draw a circle from the center of the image\n        circle_img = np.zeros((img_height, img_width), np.uint8)\n        cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n        img = cv2.bitwise_and(img, img, mask=circle_img)\n        return img\n\n    def image_preprocessing(self, img, is_cropping = True, is_gaussianblur = True):\n\n        \"\"\"\n        This function takes an image -> crops the extra dark part, resizes, draw a circle on it, and finally adds a gaussian blur to the images\n        Args : image - (numpy.ndarray) an image which we need to process\n           cropping - (boolean) whether to perform cropping of extra part(True by Default) or not(False)\n           gaussian_blur - (boolean) whether to apply gaussian blur to an image(True by Default) or not(False)\n        Output : (numpy.ndarray) preprocessed image\n        \"\"\"\n\n        if img.ndim == 2:\n            img = self.cropping_2D(img, is_cropping)  #calling cropping_2D for a GRAY image\n        else:\n            img = self.cropping_3D(img, is_cropping)  #calling cropping_3D for a RGB image\n        img = cv2.resize(img, (self.img_height, self.img_width))  # resizing the image with specified values\n        img = self.draw_circle(img)  #calling draw_circle\n        img = self.Gaussian_blur(img, is_gaussianblur) #calling Gaussian_blur\n        return img","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:35:25.640475Z","iopub.execute_input":"2023-05-21T15:35:25.640841Z","iopub.status.idle":"2023-05-21T15:35:25.660525Z","shell.execute_reply.started":"2023-05-21T15:35:25.640810Z","shell.execute_reply":"2023-05-21T15:35:25.659395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def images_per_class(class_labels,num, data_x , is_preprocess = False):\n\n    \"\"\" \n    This function plots \"num\" number of images per each class\n    Args : class_labels - (Series Object) which contains the class_labels of train or validation sets.\n           num - (Integer) sample number of images to be plot per each class\n           data_x - (Series Object) which contains the id_code of each point in train or validation sets.\n           is_preprocess - (boolean) whether to perform image processing(True) on image or not(False by Default) \n    Output : None - this function doesn't return anything.\n    \"\"\"\n\n    # class_labels num data_x data_y\n    labels = list(set(class_labels))\n    classes = ['No DR','Mild','Moderate','Severe','Proliferative DR']\n    iter=0\n    for i in labels:\n        j=1\n        plt.figure(figsize=(20,5))\n        for row in range(len(data_x)):\n            if class_labels.iloc[row] == i:\n                if is_preprocess == False:plt.subplot(1,num,j)\n                else: plt.subplot(1,num*2,j)\n                img = cv2.imread('/kaggle/input/aptos2019-blindness-detection/train_images/'+data_x.iloc[row]+'.png')\n                img1 = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n                plt.imshow(img1)\n                plt.axis('off')\n                plt.title(\"Class = {} ({})\".format(class_labels.iloc[row],classes[iter]))\n                j+=1\n                if is_preprocess == True:\n                    obj = ImageProcessing(img_width,img_height,no_channels,sigmaX=14)\n                    image = obj.image_preprocessing(img)\n                    plt.subplot(1,num*2,j)\n                    plt.imshow(image)\n                    plt.axis('off')\n                    plt.title('==> After Image Processing')\n                    j+=1\n            if is_preprocess == False and j>num: break\n            elif is_preprocess == True and j>num*2: break\n        iter+=1\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:35:30.348716Z","iopub.execute_input":"2023-05-21T15:35:30.349620Z","iopub.status.idle":"2023-05-21T15:35:30.362895Z","shell.execute_reply.started":"2023-05-21T15:35:30.349586Z","shell.execute_reply":"2023-05-21T15:35:30.361993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_per_class(train_labels,5,train_x,False)  #printing 5 random images per each class.","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:35:34.344796Z","iopub.execute_input":"2023-05-21T15:35:34.345235Z","iopub.status.idle":"2023-05-21T15:35:52.408925Z","shell.execute_reply.started":"2023-05-21T15:35:34.345197Z","shell.execute_reply":"2023-05-21T15:35:52.407302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Exploring the Dataset:","metadata":{}},{"cell_type":"code","source":"def plotting(img, title,i):\n    \"\"\"\n    This function is used for subplots\n    Args: img (numpy.ndarray) - image we need to plot\n          title(string) - title of the plot\n          i (integer) -  column number\n    output: None - this function doesn't return anything.\n    \"\"\"\n    plt.subplot(1,5,i)\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(title)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:36:16.968966Z","iopub.execute_input":"2023-05-21T15:36:16.969335Z","iopub.status.idle":"2023-05-21T15:36:16.976487Z","shell.execute_reply.started":"2023-05-21T15:36:16.969304Z","shell.execute_reply":"2023-05-21T15:36:16.975436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"obj1 = ImageProcessing(img_width,img_height, no_channels, sigmaX = 14)\nimg = '/kaggle/input/aptos2019-blindness-detection/train_images/201f882365d3.png'  #random train image\nimg = cv2.imread(img)\nimg1 = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nplt.figure(figsize=(20,5))\nplotting(img1,'Before Image Processing:',1)\nimg1 = obj1.cropping_3D(img1)\nplotting(img1,'Step-1: (Cropping extra dark pixels)',2)\nimg1 = cv2.resize(img1, (img_height,img_width))\nplotting(img1,'Step-2: (Resizing the image)',3)\nimg1 = obj1.draw_circle(img1)\nplotting(img1,'Step-3: (Drawing circle)',4)\nimg = obj1.image_preprocessing(img,'True')\nplotting(img,'Step-4: (Adding gaussian blur)',5)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:36:39.843909Z","iopub.execute_input":"2023-05-21T15:36:39.844307Z","iopub.status.idle":"2023-05-21T15:36:40.902034Z","shell.execute_reply.started":"2023-05-21T15:36:39.844274Z","shell.execute_reply":"2023-05-21T15:36:40.901242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting sample images of each class before and after Image Preprocessing:","metadata":{}},{"cell_type":"code","source":"images_per_class(train_labels,3,train_x,True)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:37:15.120465Z","iopub.execute_input":"2023-05-21T15:37:15.120857Z","iopub.status.idle":"2023-05-21T15:37:29.250234Z","shell.execute_reply.started":"2023-05-21T15:37:15.120830Z","shell.execute_reply":"2023-05-21T15:37:29.249312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image Convertion:","metadata":{}},{"cell_type":"code","source":"def image_2_vector(data, sep):\n    \"\"\"\n    This function is used for Converting an images into a vector and storing it in a file (.npy) format.\n    Input: data (Series Object) - which contains the path of the images\n           sep (String)   - used in file creation\n    Output: None - This function doesn't return anything.\n    \"\"\"\n    start_time = time.time()  # storing timestamp \n    image_vector = np.empty((len(data),img_width, img_height, no_channels), dtype = np.uint8)\n    image_processing = ImageProcessing(img_width, img_height, no_channels, sigmaX)  # Object creation\n    if sep !='test':\n        c = '/kaggle/input/aptos2019-blindness-detection/train_images/'\n    else:\n        c = '/kaggle/input/aptos2019-blindness-detection/test_images/'\n    for iter,row in enumerate(tqdm(data)): \n        img_path = c+data.iloc[iter]+'.png'\n        img = cv2.imread(img_path)\n        img = image_processing.image_preprocessing(img)    #calling image_preprocessing\n        image_vector[iter,:,:,:] = img\n\n    if sep == 'training': print(\"\\nShape of the vector:\",image_vector.shape)\n    else: print(\"\\n\\nShape of the vector:\",image_vector.shape)\n    print(\"Time taken to process the {} images: {} seconds\".format(sep,np.round(time.time()-start_time,5)))\n    path = '/kaggle/working/processed_images'\n    print(\"... Saving image_vector to {}\".format(path+'/'+sep))\n    \n    if sep == 'training': \n        print(\"\\n\",\"-\"*100,\"\\n\")\n    if not os.path.exists(path):\n        os.makedirs(path)\n    np.save(path+'/'+sep+'.npy', image_vector)  #saving file","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:38:51.577806Z","iopub.execute_input":"2023-05-21T15:38:51.578584Z","iopub.status.idle":"2023-05-21T15:38:51.590741Z","shell.execute_reply.started":"2023-05-21T15:38:51.578551Z","shell.execute_reply":"2023-05-21T15:38:51.589748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sigmaX = 14 \nimage_2_vector(train_x, \"training\") # function calling \nimage_2_vector(validation_x,\"validation\")  #function calling ","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:39:02.589034Z","iopub.execute_input":"2023-05-21T15:39:02.589454Z","iopub.status.idle":"2023-05-21T15:50:38.400825Z","shell.execute_reply.started":"2023-05-21T15:39:02.589421Z","shell.execute_reply":"2023-05-21T15:50:38.399746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/working/test.csv\") \nimage_2_vector(test['id_code'], 'test')   #function calling ","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:52:56.720834Z","iopub.execute_input":"2023-05-21T15:52:56.721803Z","iopub.status.idle":"2023-05-21T15:55:37.889528Z","shell.execute_reply.started":"2023-05-21T15:52:56.721767Z","shell.execute_reply":"2023-05-21T15:55:37.887885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = np.load('/kaggle/working/processed_images/training.npy')  #training set\nvalidation_x = np.load('/kaggle/working/processed_images/validation.npy')  #validation set\ntest_x = np.load('/kaggle/working/processed_images/test.npy')    #test set","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:56:52.017172Z","iopub.execute_input":"2023-05-21T15:56:52.017592Z","iopub.status.idle":"2023-05-21T15:57:11.621800Z","shell.execute_reply.started":"2023-05-21T15:56:52.017558Z","shell.execute_reply":"2023-05-21T15:57:11.620495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Testing:","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5)) \nplt.subplot(131)\nplt.imshow(train_x[8])   #random training example\nplt.axis('off') \nplt.title(\"Training sample image\")\nplt.subplot(132)\nplt.imshow(validation_x[120])    #random validation example\nplt.title(\"Validation sample image\")\nplt.axis('off')\nplt.subplot(133)\nplt.imshow(test_x[1200])        #random test example\nplt.title(\"Testing sample image\")\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T15:58:33.506270Z","iopub.execute_input":"2023-05-21T15:58:33.507347Z","iopub.status.idle":"2023-05-21T15:58:34.122075Z","shell.execute_reply.started":"2023-05-21T15:58:33.507308Z","shell.execute_reply":"2023-05-21T15:58:34.121034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}