{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Post Processing","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport cv2\nimport os\nimport datetime\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import DenseNet121\nimport tensorflow_addons as tfa\nfrom tensorflow.keras import layers\nimport tensorflow as tf\nfrom tqdm import notebook\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import layers\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nfrom prettytable import PrettyTable\nimport lime\nfrom lime import lime_image\nfrom skimage.segmentation import mark_boundaries","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-16T19:36:36.333916Z","iopub.execute_input":"2023-01-16T19:36:36.334282Z","iopub.status.idle":"2023-01-16T19:36:36.343100Z","shell.execute_reply.started":"2023-01-16T19:36:36.334249Z","shell.execute_reply":"2023-01-16T19:36:36.342197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 256\nBATCH_SIZE = 32","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.850595Z","iopub.execute_input":"2023-01-16T17:25:28.851535Z","iopub.status.idle":"2023-01-16T17:25:28.859522Z","shell.execute_reply.started":"2023-01-16T17:25:28.851492Z","shell.execute_reply":"2023-01-16T17:25:28.857736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading the data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.862982Z","iopub.execute_input":"2023-01-16T17:25:28.863687Z","iopub.status.idle":"2023-01-16T17:25:28.910958Z","shell.execute_reply.started":"2023-01-16T17:25:28.863641Z","shell.execute_reply":"2023-01-16T17:25:28.910052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Test Split","metadata":{}},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(df['id_code'], df['diagnosis'], test_size=0.15, stratify=df['diagnosis'],random_state=100)\nx_train = x_train.reset_index(drop=True)\nx_val = x_val.reset_index(drop=True)\ny_train = y_train.reset_index(drop=True)\ny_val = y_val.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.913274Z","iopub.execute_input":"2023-01-16T17:25:28.913524Z","iopub.status.idle":"2023-01-16T17:25:28.926864Z","shell.execute_reply.started":"2023-01-16T17:25:28.913499Z","shell.execute_reply":"2023-01-16T17:25:28.925956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"x_train = x_train.apply(lambda i:'/kaggle/input/aptos2019-blindness-detection/train_images/' + i + \".png\")\nx_val = x_val.apply(lambda i: '/kaggle/input/aptos2019-blindness-detection/train_images/' + i + \".png\")","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.928472Z","iopub.execute_input":"2023-01-16T17:25:28.928894Z","iopub.status.idle":"2023-01-16T17:25:28.936198Z","shell.execute_reply.started":"2023-01-16T17:25:28.928844Z","shell.execute_reply":"2023-01-16T17:25:28.935256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img, tol=7):\n    \"\"\"\n    Applies masks to the orignal image and \n    returns the a preprocessed image with \n    3 channels\n    \n    :param img: A NumPy Array that will be cropped\n    :param tol: The tolerance used for masking\n    \n    :return: A NumPy array containing the cropped image\n    \"\"\"\n    # For Grayscale images\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(axis = 1),mask.any(axis = 0))] # mask.any(axis = 1) makes a boolean array where each element in the array corresponds to each row in the image matrix. For a given row, the corresponding boolean value in the array is true if any value in the row is true.\n                                                                  # mask.any(axis = 0) makes a boolean array where each element in the array corresponds to each col in the image matrix. For a given col, the corresponding boolean value in the array is true if any value in the col is true.\n                                                                  # np.ix_(mask.any(axis = 1),mask.any(axis = 0)) gets those pixels from the image for which both the row and the column value is true.\n    \n    # If we have a normal RGB images\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY) #RGB image to grayscale\n        mask = gray_img > tol #creates a boolean matrix\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # Whole image is cropped as it was too dark,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(axis = 1),mask.any(axis = 0))] #applies mask to pixel 0\n            img2=img[:,:,1][np.ix_(mask.any(axis = 1),mask.any(axis = 0))]\n            img3=img[:,:,2][np.ix_(mask.any(axis = 1),mask.any(axis = 0))]\n            img = np.stack([img1,img2,img3],axis=-1)\n        return img","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.937773Z","iopub.execute_input":"2023-01-16T17:25:28.938314Z","iopub.status.idle":"2023-01-16T17:25:28.949748Z","shell.execute_reply.started":"2023-01-16T17:25:28.938280Z","shell.execute_reply":"2023-01-16T17:25:28.948933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(image_path):\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = crop_image_from_gray(image)\n    image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n    image = cv2.addWeighted (image, 4, cv2.GaussianBlur(image, (0,0) ,10), -4, 128)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.951335Z","iopub.execute_input":"2023-01-16T17:25:28.951688Z","iopub.status.idle":"2023-01-16T17:25:28.962406Z","shell.execute_reply.started":"2023-01-16T17:25:28.951630Z","shell.execute_reply":"2023-01-16T17:25:28.961370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making an array of train and validation images","metadata":{}},{"cell_type":"code","source":"train_images = np.empty((len(x_train),IMG_SIZE,IMG_SIZE,3), dtype='uint8')\nfor i,path in enumerate(notebook.tqdm(x_train)):\n    train_images[i] = preprocess(path)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:25:28.963755Z","iopub.execute_input":"2023-01-16T17:25:28.964220Z","iopub.status.idle":"2023-01-16T17:35:30.591647Z","shell.execute_reply.started":"2023-01-16T17:25:28.964185Z","shell.execute_reply":"2023-01-16T17:35:30.590348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_images = np.empty((len(x_val),IMG_SIZE,IMG_SIZE,3), dtype='uint8')\nfor i,path in enumerate(notebook.tqdm(x_val)):\n    val_images[i] = preprocess(path)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:35:30.593489Z","iopub.execute_input":"2023-01-16T17:35:30.594207Z","iopub.status.idle":"2023-01-16T17:37:19.378192Z","shell.execute_reply.started":"2023-01-16T17:35:30.594163Z","shell.execute_reply":"2023-01-16T17:37:19.377203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One hot encoding target variable","metadata":{}},{"cell_type":"code","source":"y_train = pd.get_dummies(y_train).values\ny_val = pd.get_dummies(y_val).values","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:19.381808Z","iopub.execute_input":"2023-01-16T17:37:19.382599Z","iopub.status.idle":"2023-01-16T17:37:19.393603Z","shell.execute_reply.started":"2023-01-16T17:37:19.382561Z","shell.execute_reply":"2023-01-16T17:37:19.392788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Performance Metric","metadata":{}},{"cell_type":"code","source":"qwk = tfa.metrics.CohenKappa(5,weightage='quadratic')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:19.394817Z","iopub.execute_input":"2023-01-16T17:37:19.395862Z","iopub.status.idle":"2023-01-16T17:37:22.079857Z","shell.execute_reply.started":"2023-01-16T17:37:19.395826Z","shell.execute_reply":"2023-01-16T17:37:22.079039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Callback","metadata":{}},{"cell_type":"code","source":"class Save_CB(tf.keras.callbacks.Callback):\n    def __init__(self,name):\n        self.name = name\n    def on_train_begin(self, logs={}):\n        self.max = 0\n    def on_epoch_end(self, epoch, logs={}):\n        if logs['val_cohen_kappa'] > self.max:\n            self.max = logs['val_cohen_kappa']\n            print('Validation kappa improved, saving model')\n            self.model.save(str(round(self.max,2))+'_'+self.name+'_.h5')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:22.081670Z","iopub.execute_input":"2023-01-16T17:37:22.082094Z","iopub.status.idle":"2023-01-16T17:37:22.089640Z","shell.execute_reply.started":"2023-01-16T17:37:22.082010Z","shell.execute_reply":"2023-01-16T17:37:22.087808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training (DenseNet121)","metadata":{}},{"cell_type":"code","source":"cb = Save_CB('Dense121')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:22.091502Z","iopub.execute_input":"2023-01-16T17:37:22.092416Z","iopub.status.idle":"2023-01-16T17:37:22.100168Z","shell.execute_reply.started":"2023-01-16T17:37:22.092380Z","shell.execute_reply":"2023-01-16T17:37:22.099376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"densenet = DenseNet121(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(IMG_SIZE,IMG_SIZE,3)\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:22.101574Z","iopub.execute_input":"2023-01-16T17:37:22.102269Z","iopub.status.idle":"2023-01-16T17:37:25.485567Z","shell.execute_reply.started":"2023-01-16T17:37:22.102232Z","shell.execute_reply":"2023-01-16T17:37:25.484614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = densenet.output\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dropout(0.5)(x)\ndensenet_op = layers.Dense(5, activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:25.487988Z","iopub.execute_input":"2023-01-16T17:37:25.488733Z","iopub.status.idle":"2023-01-16T17:37:25.507373Z","shell.execute_reply.started":"2023-01-16T17:37:25.488690Z","shell.execute_reply":"2023-01-16T17:37:25.506543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_dn = tf.keras.Model(inputs = densenet.inputs, outputs=densenet_op)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:25.508839Z","iopub.execute_input":"2023-01-16T17:37:25.509205Z","iopub.status.idle":"2023-01-16T17:37:25.536664Z","shell.execute_reply.started":"2023-01-16T17:37:25.509171Z","shell.execute_reply":"2023-01-16T17:37:25.535868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_dn.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:25.537804Z","iopub.execute_input":"2023-01-16T17:37:25.538481Z","iopub.status.idle":"2023-01-16T17:37:25.590660Z","shell.execute_reply.started":"2023-01-16T17:37:25.538443Z","shell.execute_reply":"2023-01-16T17:37:25.589541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_dn.compile(loss='categorical_crossentropy',\n              optimizer=tf.keras.optimizers.Adam(learning_rate=0.00005),\n              metrics=[qwk]\n              )","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:25.591499Z","iopub.execute_input":"2023-01-16T17:37:25.591799Z","iopub.status.idle":"2023-01-16T17:37:25.612923Z","shell.execute_reply.started":"2023-01-16T17:37:25.591767Z","shell.execute_reply":"2023-01-16T17:37:25.612100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dn = model_dn.fit(train_images, y_train, \n                            steps_per_epoch= len(y_train) // BATCH_SIZE, \n                            epochs=30,\n                            validation_data = (val_images,y_val), \n                            validation_steps = len(y_val) // BATCH_SIZE,\n                            callbacks = cb\n                           )","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:37:25.615636Z","iopub.execute_input":"2023-01-16T17:37:25.616419Z","iopub.status.idle":"2023-01-16T17:49:59.068427Z","shell.execute_reply.started":"2023-01-16T17:37:25.616384Z","shell.execute_reply":"2023-01-16T17:49:59.067402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = tf.keras.models.load_model('/kaggle/working/0.9_Dense121_.h5')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:52:57.262512Z","iopub.execute_input":"2023-01-16T17:52:57.262970Z","iopub.status.idle":"2023-01-16T17:53:01.160075Z","shell.execute_reply.started":"2023-01-16T17:52:57.262931Z","shell.execute_reply":"2023-01-16T17:53:01.158964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = final.predict(val_images)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:53:08.304529Z","iopub.execute_input":"2023-01-16T17:53:08.304884Z","iopub.status.idle":"2023-01-16T17:53:12.356774Z","shell.execute_reply.started":"2023-01-16T17:53:08.304852Z","shell.execute_reply":"2023-01-16T17:53:12.355764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Finding absolute errors","metadata":{}},{"cell_type":"code","source":"dif = y_val-y_pred\nerrors = np.amax(dif, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T17:56:30.792121Z","iopub.execute_input":"2023-01-16T17:56:30.793149Z","iopub.status.idle":"2023-01-16T17:56:30.798207Z","shell.execute_reply.started":"2023-01-16T17:56:30.793077Z","shell.execute_reply":"2023-01-16T17:56:30.796852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Postprocessing","metadata":{}},{"cell_type":"markdown","source":"## Distribution of absolute errors","metadata":{}},{"cell_type":"code","source":"sns.histplot(x = errors, kde=True, binwidth=0.1)\nplt.ylabel('Frequency')\nplt.xlabel('Absolute Error Value')\nplt.title('Distribution Of Errors Of Each Prediction')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:04:53.303596Z","iopub.execute_input":"2023-01-16T18:04:53.303970Z","iopub.status.idle":"2023-01-16T18:04:53.524010Z","shell.execute_reply.started":"2023-01-16T18:04:53.303938Z","shell.execute_reply":"2023-01-16T18:04:53.523104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We can see that as the error value increases, the error decreases, which is a good sign\n* Most of the points lie in the range of 0 to 0.1\n* There is an increase in the number of points which have an absolute error of 0.8 to 1\n* These images are causing the maximum problem to the model and are being misclassified ","metadata":{}},{"cell_type":"markdown","source":"## Dividing the data into good, medium and bad","metadata":{}},{"cell_type":"markdown","source":"* Points whose absolute error is less than 1/3 are categorized as good\n* Points whose absolute error is between 1/3 and 2/3 are categorized as medium\n* Points whose absolute error is greater than 2/3 are categorized as bad","metadata":{}},{"cell_type":"code","source":"good = []\nmedium = []\nbad = []\nfor index,error in enumerate(errors):\n    if error < 1/3:\n        good.append(index)\n    elif error < 2/3:\n        medium.append(index)\n    else:\n        bad.append(index)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:23:32.048449Z","iopub.execute_input":"2023-01-16T18:23:32.048821Z","iopub.status.idle":"2023-01-16T18:23:32.056008Z","shell.execute_reply.started":"2023-01-16T18:23:32.048779Z","shell.execute_reply":"2023-01-16T18:23:32.054838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of points in good, medium and bad arrays","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 5))\nsns.barplot(x=['good','medium','bad'],y=[len(good),len(medium),len(bad)],palette='Accent',ax=ax)\nfor p in ax.patches:\n   ax.annotate('{}'.format(int(p.get_height())), (p.get_x(), p.get_height()))\nplt.ylabel('Count')\nplt.xlabel('Category')\nplt.title('Count of points in each class')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:27:40.009842Z","iopub.execute_input":"2023-01-16T18:27:40.010222Z","iopub.status.idle":"2023-01-16T18:27:40.191568Z","shell.execute_reply.started":"2023-01-16T18:27:40.010190Z","shell.execute_reply":"2023-01-16T18:27:40.190600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Good Category Images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (13,13))\nfor i,x in enumerate(val_images[good][5:10]):\n    plt.subplot(1,5,i+1)\n    plt.axis('off')\n    plt.imshow(x)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:38:41.220553Z","iopub.execute_input":"2023-01-16T18:38:41.220920Z","iopub.status.idle":"2023-01-16T18:38:41.572020Z","shell.execute_reply.started":"2023-01-16T18:38:41.220887Z","shell.execute_reply":"2023-01-16T18:38:41.571129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Medium Category Images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (13,13))\nfor i,x in enumerate(val_images[medium][5:10]):\n    plt.subplot(1,5,i+1)\n    plt.axis('off')\n    plt.imshow(x)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:38:26.346845Z","iopub.execute_input":"2023-01-16T18:38:26.347216Z","iopub.status.idle":"2023-01-16T18:38:26.674794Z","shell.execute_reply.started":"2023-01-16T18:38:26.347183Z","shell.execute_reply":"2023-01-16T18:38:26.673961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bad Category Images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (13,13))\nfor i,x in enumerate(val_images[bad][5:10]):\n    plt.subplot(1,5,i+1)\n    plt.axis('off')\n    plt.imshow(x)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:38:13.520288Z","iopub.execute_input":"2023-01-16T18:38:13.520739Z","iopub.status.idle":"2023-01-16T18:38:14.047935Z","shell.execute_reply.started":"2023-01-16T18:38:13.520704Z","shell.execute_reply":"2023-01-16T18:38:14.047042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The model works well with eye images with very little to no spots\n* The model shows medium performance on eye images with spots that are concentrated in a single place\n* The model gets a bad scorew with those eye images, that have scattered spots\n* Most of the images belonging to the medium and bad categories don't have the complete eyeball in the image","metadata":{}},{"cell_type":"markdown","source":"## Distribution of classes in good medium and bad categories","metadata":{}},{"cell_type":"code","source":"sns.countplot(x = np.argmax(y_val[good], axis=1) ,palette='Accent')\nplt.xlabel('Class')\nplt.ylabel('Count')\nplt.title(\"Distribution of classes in category good\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:53:23.035255Z","iopub.execute_input":"2023-01-16T18:53:23.035780Z","iopub.status.idle":"2023-01-16T18:53:23.221132Z","shell.execute_reply.started":"2023-01-16T18:53:23.035746Z","shell.execute_reply":"2023-01-16T18:53:23.220161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = np.argmax(y_val[medium], axis=1) ,palette='Accent')\nplt.xlabel('Class')\nplt.ylabel('Count')\nplt.title(\"Distribution of classes in category medium\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T18:53:41.482184Z","iopub.execute_input":"2023-01-16T18:53:41.482860Z","iopub.status.idle":"2023-01-16T18:53:41.674885Z","shell.execute_reply.started":"2023-01-16T18:53:41.482826Z","shell.execute_reply":"2023-01-16T18:53:41.673982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = np.argmax(y_val[bad], axis=1) ,palette='Accent')\nplt.xlabel('Class')\nplt.ylabel('Count')\nplt.title(\"Distribution of classes in category bad\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T19:08:48.760178Z","iopub.execute_input":"2023-01-16T19:08:48.764413Z","iopub.status.idle":"2023-01-16T19:08:48.920204Z","shell.execute_reply.started":"2023-01-16T19:08:48.764363Z","shell.execute_reply":"2023-01-16T19:08:48.919355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The model faces difficulty in classifying points of class 1 and class 4\n* Identifying class 1 images is especially difficult because finding small spots in the eye images is not an easy task even for humans ","metadata":{}},{"cell_type":"markdown","source":"## Using LIME to see what regions of the image help in classification","metadata":{}},{"cell_type":"markdown","source":"* Green regions are the pixels that boost the probability of the image belonging to the predicted class\n* Red regions are the pixels that reduce the probability of the image belonging to the predicted class","metadata":{}},{"cell_type":"markdown","source":"### Good Category","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (13,13))\nfor x in range(5):\n    explanation = explainer.explain_instance(val_images[good][x], final.predict,  \n                                         top_labels=3, hide_color=0, num_samples=1000) #setting up explainer model\n    temp_1, mask_1 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=False, num_features=10, hide_rest=False) #getting the image and mask\n    plt.subplot(2,5,x+1)\n    plt.imshow(val_images[good][x])\n    plt.axis('off')\n    plt.subplot(2,5,x+6)\n    plt.imshow(mark_boundaries(temp_1, mask_1))\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T19:45:10.268051Z","iopub.execute_input":"2023-01-16T19:45:10.268561Z","iopub.status.idle":"2023-01-16T19:46:07.644791Z","shell.execute_reply.started":"2023-01-16T19:45:10.268527Z","shell.execute_reply":"2023-01-16T19:46:07.643967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Medium Category","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (13,13))\nfor x in range(5):\n    explanation = explainer.explain_instance(val_images[medium][x], final.predict,  \n                                         top_labels=3, hide_color=0, num_samples=1000)\n    temp_1, mask_1 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=False, num_features=10, hide_rest=False)\n    plt.subplot(2,5,x+1)\n    plt.imshow(val_images[medium][x])\n    plt.axis('off')\n    plt.subplot(2,5,x+6)\n    plt.imshow(mark_boundaries(temp_1, mask_1))\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T19:46:43.541496Z","iopub.execute_input":"2023-01-16T19:46:43.541908Z","iopub.status.idle":"2023-01-16T19:47:39.560047Z","shell.execute_reply.started":"2023-01-16T19:46:43.541873Z","shell.execute_reply":"2023-01-16T19:47:39.559148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bad Category","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (13,13))\nfor x in range(5):\n    explanation = explainer.explain_instance(val_images[bad][x], final.predict,  \n                                         top_labels=3, hide_color=0, num_samples=1000)\n    temp_1, mask_1 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=False, num_features=10, hide_rest=False)\n    plt.subplot(2,5,x+1)\n    plt.imshow(val_images[bad][x])\n    plt.axis('off')\n    plt.subplot(2,5,x+6)\n    plt.imshow(mark_boundaries(temp_1, mask_1))\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-01-16T19:47:39.562122Z","iopub.execute_input":"2023-01-16T19:47:39.562913Z","iopub.status.idle":"2023-01-16T19:48:36.903054Z","shell.execute_reply.started":"2023-01-16T19:47:39.562874Z","shell.execute_reply":"2023-01-16T19:48:36.902248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* As the absolute error increases, the size of the green region increases and the size of the red region decreases.\n* Images that belong to the good class show a neat segregation between the border and the eyeball\n* Absolute error is more in images where the model is unable to identify the eyeball","metadata":{}}]}