{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":29282,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-27T13:11:08.999787Z","iopub.execute_input":"2023-11-27T13:11:09.000330Z","iopub.status.idle":"2023-11-27T13:11:09.257029Z","shell.execute_reply.started":"2023-11-27T13:11:09.000103Z","shell.execute_reply":"2023-11-27T13:11:09.256237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, sys\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport skimage.io\nfrom skimage.transform import resize\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport PIL\nfrom PIL import Image, ImageOps\nimport cv2\nfrom sklearn.utils import class_weight, shuffle\nfrom keras.losses import binary_crossentropy\nfrom keras.applications.resnet50 import preprocess_input\nimport keras.backend as K\nimport tensorflow as tf\nfrom sklearn.metrics import f1_score, fbeta_score\nfrom keras.utils import Sequence\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\n\nWORKERS = 2\nCHANNEL = 3\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nIMG_SIZE = 512\nNUM_CLASSES = 5\nSEED = 77\nTRAIN_NUM = 1000 # use 1000 when you just want to explore new idea, use -1 for full train","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-27T13:11:17.680660Z","iopub.execute_input":"2023-11-27T13:11:17.681253Z","iopub.status.idle":"2023-11-27T13:11:21.819676Z","shell.execute_reply.started":"2023-11-27T13:11:17.681180Z","shell.execute_reply":"2023-11-27T13:11:21.818787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UPDATE on V9:\nThis kernel have two important updates.\n\n* Before Version 8, I couldn't make Ben's and Cropping method work together nicely, so I emphasized on gray scale. Now, I adjust both functions and beleive that color version is better than gray scale.\n\n* Before Version 9, I found a bug that will cause an old crop function to fail in a private test set (it works fine on training and public test sets). Here, I fix that bug. However, I still cannot guarantee whether there will be any more cases on private test set that will fail the crop function. **Update on V11** Now I was able to have a valid LB score with the new crop function, so if anybody still have some submission errors, that is the reason of other bugs.\n\n## update on V14.\n* Compare to circle crop in Section 3.A2 according to @taindow : please visit his kernel : https://www.kaggle.com/taindow/pre-processing-train-and-test-images\n\nOther minor updates. Note on estimation inconsistency and Aravind's history.","metadata":{}},{"cell_type":"markdown","source":"\n# 1. Introduction. Explore first, train later.\n\nHi everyone! As *Aravind Eye Hospital* is one of my favorite organization in the world; they take care of poor people's eyes for free with an impressive sustainable business model.  I will try my best to contribute something to our community. One intuitive way to improve the performance of our model is to simply improve the quality of input images. In this kernel, I will share two ideas which I hope may be useful to some of you : \n\n- **Reducing lighting-condition effects** : as we will see, images come with many different lighting conditions, some images are very dark and difficult to visualize. We can try to convert the image to gray scale, and visualize better. Alternatively, there is a better approach. We can try the method of [Ben Graham (last competition's winner)](https://github.com/btgraham/SparseConvNet/tree/kaggle_Diabetic_Retinopathy_competition)\n- **Cropping uninformative area** : everyone know this :) Here, I just find the codes from internet and choose the best one for you :)\n\nWe are going to apply both techniques to both the official data, and the past competition data (shout out @tanlikesmath for creating this dataset! https://www.kaggle.com/tanlikesmath/diabetic-retinopathy-resized . In the updated version, I also try @donkeys' dataset https://www.kaggle.com/donkeys/retinopathy-train-2015 , which is .png which may be have higer image quality than .jpeg format)\n\nIf I found more useful tricks, I will update the notebook, or if you have more useful tricks and would love to share, please let me know!\n\nI use some parts of codes from @mathormad and @artgor kernels. Thanks both of you!","metadata":{}},{"cell_type":"markdown","source":"Now let us start by loading the train/test dataframes. The `train_test_split` here is in fact not necessary. But when I first fork the kernel from @mathormad, I found some interesting examples using this split and the current `SEED`, so I continue to use them here.","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ndf_test = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\n\nx = df_train['id_code']\ny = df_train['diagnosis']\n\nx, y = shuffle(x, y, random_state=SEED)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:11:30.407280Z","iopub.execute_input":"2023-11-27T13:11:30.407659Z","iopub.status.idle":"2023-11-27T13:11:30.455971Z","shell.execute_reply.started":"2023-11-27T13:11:30.407590Z","shell.execute_reply":"2023-11-27T13:11:30.455143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x, valid_x, train_y, valid_y = train_test_split(x, y, test_size=0.15,\n                                                      stratify=y, random_state=SEED)\nprint(train_x.shape, train_y.shape, valid_x.shape, valid_y.shape)\ntrain_y.hist()\nvalid_y.hist()","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:11:35.185234Z","iopub.execute_input":"2023-11-27T13:11:35.185596Z","iopub.status.idle":"2023-11-27T13:11:35.608505Z","shell.execute_reply.started":"2023-11-27T13:11:35.185536Z","shell.execute_reply":"2023-11-27T13:11:35.607218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.1 Simple picture to explain Diabetic Retinopathy\n\nHow do we know that a patient have diabetic retinopahy? **[There are at least 5 things to spot on](https://www.eyeops.com/contents/our-services/eye-diseases/diabetic-retinopathy)**. Image credit https://www.eyeops.com/\n![credit : https://www.eyeops.com/](https://sa1s3optim.patientpop.com/assets/images/provider/photos/1947516.jpeg)","metadata":{}},{"cell_type":"markdown","source":"From quick investigations of the data (see various pictures below), I found that *Hemorrphages, Hard Exudates and Cotton Wool spots* are quite easily observed. However, I still could not find examples of *Aneurysm* or *Abnormal Growth of Blood Vessels* from our data yet. Perhaps the latter two cases are important if we want to catch up human benchmnark using our model.","metadata":{}},{"cell_type":"markdown","source":"## 1.2 Original Inputs\n\nFirst, let have a glance of original inputs. Each row depicts each severity level. We can see two problems which make the severity difficult to spot on. First, some images are very dark [pic(0,2) and pic(4,4) ] and sometimes different color illumination is confusing [pic (3,3)]. Second, we can get the uninformative dark areas for some pictures [pic(0,1), pic(0,3)]. This is important when we reduce the picture size, as informative areas become too small. So it is intuitive to crop the uninformative areas out in the second case.","metadata":{}},{"cell_type":"code","source":"%%time\nfig = plt.figure(figsize=(25, 16))\n# display 10 images from each class\nfor class_id in sorted(train_y.unique()):\n    for i, (idx, row) in enumerate(df_train.loc[df_train['diagnosis'] == class_id].sample(5, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, 5, class_id * 5 + i + 1, xticks=[], yticks=[])\n        path=f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n\n        plt.imshow(image)\n        ax.set_title('Label: %d-%d' % (class_id, idx) )\n        \n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-27T13:13:36.588056Z","iopub.execute_input":"2023-11-27T13:13:36.588451Z","iopub.status.idle":"2023-11-27T13:13:41.638593Z","shell.execute_reply.started":"2023-11-27T13:13:36.588385Z","shell.execute_reply":"2023-11-27T13:13:41.637518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#CROP FUNCTIONS\n\ndef crop_black(img,\n               tol = 7):\n\n    '''\n    Perform automatic crop of black areas\n    '''\n\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img > tol\n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n\n        if (check_shape == 0):\n            return img\n        else:\n            img1 = img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2 = img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3 = img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n            img  = np.stack([img1, img2, img3], axis = -1)\n            return img","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:16:13.353639Z","iopub.execute_input":"2023-11-27T13:16:13.354179Z","iopub.status.idle":"2023-11-27T13:16:13.364390Z","shell.execute_reply.started":"2023-11-27T13:16:13.354109Z","shell.execute_reply":"2023-11-27T13:16:13.363362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def random_crop(img,\n                size = (0.9, 1)):\n\n    '''\n    Random crop\n    '''\n\n    height, width, depth = img.shape\n\n    cut = 1 - random.uniform(size[0], size[1])\n\n    i = random.randint(0, int(cut * height))\n    j = random.randint(0, int(cut * width))\n    h = i + int((1 - cut) * height)\n    w = j + int((1 - cut) * width)\n\n    img = img[i:h, j:w, :]\n\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:17:51.545517Z","iopub.execute_input":"2023-11-27T13:17:51.545906Z","iopub.status.idle":"2023-11-27T13:17:51.553439Z","shell.execute_reply.started":"2023-11-27T13:17:51.545850Z","shell.execute_reply":"2023-11-27T13:17:51.552200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can try gray scale and feel understand better for some pictures, as color distraction is gone. For example, we can see more blood clearer in the upper part of pic(4,4), which has severity of level 4.","metadata":{}},{"cell_type":"code","source":"%%time\nfig = plt.figure(figsize=(25, 16))\n# display 10 images from each class\nfor class_id in sorted(train_y.unique()):\n    for i, (idx, row) in enumerate(df_train.loc[df_train['diagnosis'] == class_id].sample(5, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, 5, class_id * 5 + i + 1, xticks=[], yticks=[])\n        path=f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image = crop_black(image, tol = 7)\n        \n\n        plt.imshow(image)\n        ax.set_title('Label: %d-%d' % (class_id, idx) )\n        \n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:17:11.947739Z","iopub.execute_input":"2023-11-27T13:17:11.948178Z","iopub.status.idle":"2023-11-27T13:17:16.949072Z","shell.execute_reply.started":"2023-11-27T13:17:11.948108Z","shell.execute_reply":"2023-11-27T13:17:16.948052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfig = plt.figure(figsize=(25, 16))\n# display 10 images from each class\nfor class_id in sorted(train_y.unique()):\n    for i, (idx, row) in enumerate(df_train.loc[df_train['diagnosis'] == class_id].sample(5, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, 5, class_id * 5 + i + 1, xticks=[], yticks=[])\n        path=f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image = crop_black(image, tol = 7)\n        do_random_crop = False\n        \n        if do_random_crop == True:\n            image = random_crop(image, size = (0.9, 1))\n        \n\n        plt.imshow(image)\n        ax.set_title('Label: %d-%d' % (class_id, idx) )\n        \n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:18:40.714056Z","iopub.execute_input":"2023-11-27T13:18:40.714415Z","iopub.status.idle":"2023-11-27T13:18:45.520547Z","shell.execute_reply.started":"2023-11-27T13:18:40.714364Z","shell.execute_reply":"2023-11-27T13:18:45.517306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfig = plt.figure(figsize=(25, 16))\n# display 10 images from each class\nfor class_id in sorted(train_y.unique()):\n    for i, (idx, row) in enumerate(df_train.loc[df_train['diagnosis'] == class_id].sample(5, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, 5, class_id * 5 + i + 1, xticks=[], yticks=[])\n        path=f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image = crop_black(image, tol = 7)\n        do_random_crop = False\n        \n        if do_random_crop == True:\n            image = random_crop(image, size = (0.9, 1))\n        \n        image = cv2.resize(image, (int(256), int(256)))\n        image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), 25.6), -4, 128)\n        \n\n        plt.imshow(image)\n        ax.set_title('Label: %d-%d' % (class_id, idx) )\n        \n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:19:40.386182Z","iopub.execute_input":"2023-11-27T13:19:40.386540Z","iopub.status.idle":"2023-11-27T13:19:46.732621Z","shell.execute_reply.started":"2023-11-27T13:19:40.386486Z","shell.execute_reply":"2023-11-27T13:19:46.731825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def circle_crop(img,\n                sigmaX = 25.6):\n\n    '''\n    Perform circular crop around image center\n    '''\n\n    height, width, depth = img.shape\n\n    largest_side = np.max((height, width))\n    img = cv2.resize(img, (largest_side, largest_side))\n\n    height, width, depth = img.shape\n\n    x = int(width / 2)\n    y = int(height / 2)\n    r = np.amin((x,y))\n\n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness = -1)\n\n    img = cv2.bitwise_and(img, img, mask = circle_img)\n    return img","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:20:06.049801Z","iopub.execute_input":"2023-11-27T13:20:06.050431Z","iopub.status.idle":"2023-11-27T13:20:06.059425Z","shell.execute_reply.started":"2023-11-27T13:20:06.050349Z","shell.execute_reply":"2023-11-27T13:20:06.058218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfig = plt.figure(figsize=(25, 16))\n# display 10 images from each class\nfor class_id in sorted(train_y.unique()):\n    for i, (idx, row) in enumerate(df_train.loc[df_train['diagnosis'] == class_id].sample(5, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, 5, class_id * 5 + i + 1, xticks=[], yticks=[])\n        path=f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (256, 256))\n        image = crop_black(image, tol = 7)\n        do_random_crop = False\n        \n        if do_random_crop == True:\n            image = random_crop(image, size = (0.9, 1))\n        \n        image = cv2.resize(image, (int(256), int(256)))\n        image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), 25.6), -4, 128)\n        \n        image = circle_crop(image, sigmaX = 25.6)\n        \n\n        plt.imshow(image)\n        ax.set_title('Label: %d-%d' % (class_id, idx) )\n        \n","metadata":{"execution":{"iopub.status.busy":"2023-11-27T13:20:43.689615Z","iopub.execute_input":"2023-11-27T13:20:43.690010Z","iopub.status.idle":"2023-11-27T13:20:49.946435Z","shell.execute_reply.started":"2023-11-27T13:20:43.689956Z","shell.execute_reply":"2023-11-27T13:20:49.945588Z"},"trusted":true},"execution_count":null,"outputs":[]}]}