{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"https://arxiv.org/pdf/1506.02158v6.pdf\nhttps://arxiv.org/abs/1502.03167\nhttp://www.jmlr.org/papers/volume13/bergstra12a/bergstra12a.pdf\nhttps://speakerdeck.com/tmls/keras-by-keisuke-kamataki-tmls-number-2?slide=19\nhttps://arxiv.org/abs/1206.5533\nhttps://ieeexplore.ieee.org/stamp/stamp.jsp?tp=&arnumber=1372173","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# imports\nimport pandas as pd\nfrom keras.utils import np_utils\nfrom PIL import Image\nfrom keras.preprocessing import image\nimport os\nimport numpy as np\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.utils import shuffle\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPooling2D, BatchNormalization\nfrom keras.preprocessing.image import ImageDataGenerator, array_to_img, img_to_array, load_img\nimport cv2","execution_count":1,"outputs":[{"output_type":"stream","text":"Using TensorFlow backend.\n","name":"stderr"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"https://www.sciencedirect.com/science/article/pii/S2210832714000076\nhttps://www.hindawi.com/journals/jhe/2018/6275435/\nhttps://www.researchgate.net/publication/318308371_Medical_imbalanced_data_classification\nhttps://devmesh.intel.com/projects/deep-learning-for-analysis-of-imbalanced-medical-image-datasets\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# read train images\ntrainLabels = pd.read_csv(\"../input/trainLabels.csv\")\ntrainLabels.head()","execution_count":2,"outputs":[{"output_type":"execute_result","execution_count":2,"data":{"text/plain":"      image  level\n0   10_left      0\n1  10_right      0\n2   13_left      0\n3  13_right      0\n4   15_left      1","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image</th>\n      <th>level</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>10_left</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>10_right</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>13_left</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>13_right</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>15_left</td>\n      <td>1</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"metadata":{},"cell_type":"markdown","source":"#### Class \n* 0 - No DR\n* 1 - Mild\n* 2 - Moderate\n* 3 - Severe\n* 4 - Proliferative DR"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n#img_names = []\nimg_labels = []\ndef read_train_dataset():\n    img_matrix = []\n    imgs = os.listdir(\"../input\") \n    imgs.remove(\"trainLabels.csv\") \n    for item in imgs:\n        base = os.path.basename(\"../input/\" + item)\n        filename = os.path.splitext(base)[0]\n        img = load_img(\"../input/\" + item)\n        img_array = img_to_array(img)\n        img=img.resize((256,256))\n        #img_array = cv2.resize(img_array, (256, 256))\n        #img_array = np.array(img_array, dtype='float')/255.0\n        img_matrix.append(img_array)\n        img_labels.append(trainLabels.loc[trainLabels.image==filename, 'level'].values[0])\n    return np.asarray(img_matrix, dtype='float')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s = read_train_dataset()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(s, img_labels, test_size=0.33, random_state=42)\nprint('train IDs: ', len(x_train))\nprint('test IDs: ', len(x_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weights = {0:0.24, 1:1.21, 2:6.42, 3:3.05, 4:6.92}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for file in listing:\n    base = os.path.basename(\"../input/\" + file)\n    fileName = os.path.splitext(base)[0]\n    imlabel.append(trainLabels.loc[trainLabels.image==fileName, 'level'].values[0])\n    im = Image.open(\"../input/\" + file)\n    img = np.array(im.resize((img_rows,img_cols)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = []\ndata = []\nimgs = os.listdir(\"../input\") \nimgs.remove(\"trainLabels.csv\") \nfor item in imgs:\n    base = os.path.basename(\"../input/\" + item)\n    filename = os.path.splitext(base)[0]\n    labels.append(trainLabels.loc[trainLabels.image==filename, 'level'].values[0])\n    data.append(filename)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1./255)\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_directory(\n        '../input/',\n        target_size=(512, 512),\n        batch_size=32,\n        class_mode='categorical')\n\ntest_generator = test_datagen.flow_from_directory(\n        '../input/',\n        target_size=(512, 512),\n        batch_size=32,\n        class_mode='categorical')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xtrain, xtest, ytrain, ytest = train_test_split(data, labels, test_size=0.33, random_state=42)\nprint('train IDs: ', len(xtrain))\nprint('test IDs: ', len(xtest))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfrom skimage.io import imread\nfrom skimage.transform import resize\nimport numpy as np\nfrom keras.utils import Sequence\n\nclass My_Generator(Sequence):\n\n    def __init__(self, image_filenames, labels, batch_size):\n        self.image_filenames, self.labels = image_filenames, labels\n        self.batch_size = batch_size\n\n    def __len__(self):\n        return np.ceil(len(self.image_filenames) / float(self.batch_size))\n\n    def __getitem__(self, idx):\n        batch_x = self.image_filenames[idx * self.batch_size:(idx + 1) * self.batch_size]\n        batch_y = self.labels[idx * self.batch_size:(idx + 1) * self.batch_size]\n\n        return np.array([\n            resize(imread(file_name), (200, 200))\n               for file_name in batch_x]), np.array(batch_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nmy_training_batch_generator = My_Generator(xtrain, ytrain, 64)\nmy_validation_batch_generator = My_Generator(xtest, ytest, 64)\n\nmodel.fit_generator(\n        my_training_batch_generator,\n        steps_per_epoch=2000,\n        epochs=50,\n        validation_data=my_validation_batch_generator,\n        validation_steps=800)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\ntrain_generator = train_datagen.flow_from_directory()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xtrain, xtest, ytrain, ytest = train_test_split(data, labels, test_size=0.33, random_state=42)\nprint('train IDs: ', len(xtrain))\nprint('test IDs: ', len(xtest))\n\nparams = {\n    'batch_size': 64,\n    'n_channels': 3,\n    'shuffle': True\n}\n\ntraining_generator = DataGenerator(xtrain, labels, **params)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def generate_data(directory, batch_size):\n    \"\"\"Replaces Keras' native ImageDataGenerator.\"\"\"\n    i = 0\n    file_list = os.listdir(directory)\n    while True:\n        image_batch = []\n        for b in range(batch_size):\n            if i == len(file_list):\n                i = 0\n                random.shuffle(file_list)\n            sample = file_list[i]\n            i += 1\n            #image = cv2.resize(cv2.imread(sample[0]), INPUT_SHAPE)\n            image = cv2.imread(sample[0], INPUT_SHAPE)\n            image_batch.append((image.astype(float) - 128) / 128)\n\n        yield np.array(image_batch)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.compile(optimizer = keras.optimizers.SGD(lr=0.001, momentum=0.9, nesterov=True))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"model.fit_generator(\n    generate_data('../input/', 64),\n    steps_per_epoch=len(os.listdir('../input/')) // 64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get images from training set\n\n# resize the image to (256, 256)\nimg_rows, img_cols = 512, 512\n\nlisting = os.listdir(\"../input\") \nlisting.remove(\"trainLabels.csv\")\n#listing.remove(['trainLabels.csv', 'sampleSubmission.csv', 'sample.zip'])\n\nimmatrix = []\nimlabel = []\n\nfor file in listing:\n    base = os.path.basename(\"../input/\" + file)\n    fileName = os.path.splitext(base)[0]\n    imlabel.append(trainLabels.loc[trainLabels.image==fileName, 'level'].values[0])\n    im = Image.open(\"../input/\" + file)\n    img = np.array(im.resize((img_rows,img_cols)))\n    \n    # convert to green channel only\n    img[:,:,[0,2]] = 0\n    immatrix.append(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get images from training set\n\n# resize the image to (256, 256)\nimg_rows, img_cols = 512, 512\n\nlisting = os.listdir(\"../input\") \nlisting.remove(\"trainLabels.csv\")\n#listing.remove(['trainLabels.csv', 'sampleSubmission.csv', 'sample.zip'])\n\nimmatrix = []\nimlabel = []\n\nfor file in listing:\n    base = os.path.basename(\"../input/\" + file)\n    filename = os.path.splitext(base)[0]\n    imlabel.append(trainLabels.loc[trainLabels.image==filename, 'level'].values[0])\n    #im = Image.open(\"../input/\" + file)\n    #img = np.array(im.resize((img_rows,img_cols)))\n    image = load_img(\"../input/\"+file)\n    immatrix.append(img_to_array(image))\n    \n    # convert to green channel only\n    #img[:,:,[0,2]] = 0\n    #immatrix.append(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"#cv2.imread(\"../input/\"+listing[2])\na=Image.open(\"../input/\" + file)\nnp.array(a)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import img_to_array, load_img\nimport cv2\nimmatrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"im = Image.fromarray(immatrix[1],'RGB')\nprint(\"level:\",imlabel[1])\nim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# flip horizontal\n\nimg_gen = ImageDataGenerator()\nim2 = img_gen.apply_transform(immatrix[1], {'flip_horizontal':True})\nim22 = Image.fromarray(im2, 'RGB')\nim22","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"'''# image transformations\nimport random\n\n# define transformation methods\ndef horizontal_flip(image_array):\n    return image_array[:, ::-1]\n\ndef vertical_flip(image_array):\n    return image_array[::-1,:]\n\ndef random_transform(image_array):\n    if random.random() < 0.5:\n        return vertical_flip(image_array)\n    else:\n        return horizontal_flip(image_array)'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"'''im = Image.fromarray(vertical_flip(immatrix[1]),'RGB')\nim'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"'''length = len(immatrix)\nfor i in range(length):\n    if random.random() < 0.1:\n        immatrix.append(random_transform(immatrix[i]))\n        imlabel.append(imlabel[i])\n        \nprint(\"Size of image array before augmentation: \", length)\nprint(\"Size fo image array after augmentation: \", len(immatrix))'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plot class distribution\n\nplt.hist(img_labels)\nplt.show()\nCounter(img_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_matrix, img_labels = shuffle(img_matrix, img_labels, random_state=42)\n#train_data = [data,label]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(img_matrix, img_labels, test_size = 0.1, random_state = 42)\n\nprint(np.array(x_train).shape)\nprint(np.array(y_train).shape)\n\nprint(np.array(x_test).shape)\nprint(np.array(y_test).shape)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"collapsed":true},"cell_type":"code","source":"im = Image.fromarray(img_matrix[1],'RGB')\nprint(\"level:\",img_labels[0])\nim","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"collapsed":true},"cell_type":"code","source":"imgs = os.listdir(\"../input\") \nimgs.remove(\"trainLabels.csv\")\nimg = load_img(\"../input/\" + imgs[1])\nimg_array = img_to_array(img)\nimg_array = cv2.resize(img_array, (512, 512))\n#img_array = np.array(img_array, dtype='float')/255.0\nim = Image.fromarray(img_array,'RGB')\nimg","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# shuffle dataset\n\ndata,label = shuffle(immatrix, imlabel, random_state=42)\ntrain_data = [data,label]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# count per classes\nCounter(label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true,"collapsed":true},"cell_type":"code","source":"# plot class distribution\n\nplt.hist(label)\nplt.show()\nCounter(label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# split into training set and test set\n\nx_train, x_test, y_train, y_test = train_test_split(train_data[0], train_data[1], test_size = 0.1, random_state = 42)\n\nprint(np.array(x_train).shape)\nprint(np.array(y_train).shape)\n\nprint(np.array(x_test).shape)\nprint(np.array(y_test).shape)\n#https://www.kaggle.com/kevinyang372/diabetic-retinopathy-detection-cnn","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"'''# prepare data augmentation configuration\ntrain_datagen = ImageDataGenerator(\n    rescale=1. / 255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True)\n\ndatagen = ImageDataGenerator(\n        rotation_range=40,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True,\n        fill_mode='nearest')\n\nvalid_generator = valid_datagen.flow_from_directory(\n    directory=r\"./valid/\",\n    target_size=(224, 224),\n    color_mode=\"rgb\",\n    batch_size=32,\n    class_mode=\"categorical\",\n    shuffle=True,\n    seed=42\n)'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# convert labels to binary class matrix\ny_train = np_utils.to_categorical(np.array(y_train), 5)\ny_test = np_utils.to_categorical(np.array(y_test), 5)\n\nx_train = np.array(x_train).astype(\"float32\")/255.0\nx_test = np.array(x_test).astype(\"float32\")/255.0\n\nprint(np.array(y_train).shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.utils import class_weight\nclass_weights = class_weight.compute_class_weight('balanced', \n                                                 np.unique(y_train),\n                                                 y_train)\nclass_weights\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weights = {0:0.24, 1:1.21, 2:6.42, 3:3.05, 4:6.92}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# build cnn\nmodel = Sequential()\n\nmodel.add(Conv2D(16, 11, padding='same', activation='relu', input_shape=x_train[0].shape, use_bias=False))\n#model.add(Conv2D(16, 11, activation='relu'))\nmodel.add(MaxPooling2D())\n#model.add(Dropout(0.25))\n\nmodel.add(Conv2D(32, 9, padding='same', activation='relu', use_bias=False))\n#model.add(Conv2D(32, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D())\n#model.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, 7, padding='same', activation='relu', use_bias=False))\n#model.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D())\n#model.add(Dropout(0.25))\n\n#model.add(Conv2D(128, 3, padding='same', activation='relu', use_bias=False))\n#model.add(MaxPooling2D())\n#model.add(Dropout(0.25))\n\n#model.add(Conv2D(256, 3, padding='same', activation='relu', use_bias=False))\n#model.add(MaxPooling2D())\n\n#model.add(Conv2D(512, 3, padding='same', activation='relu', use_bias=False))\n#model.add(MaxPooling2D())\n\n#model.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='sigmoid', use_bias=False))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(activation='softmax', units=5))\n\nmodel.compile(loss='categorical_crossentropy', optimizer = keras.optimizers.SGD(lr=0.001, momentum=0.9, nesterov=True), metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"# add some visualization\nfrom IPython.display import SVG\nfrom keras.utils.vis_utils import model_to_dot\nSVG(model_to_dot(model).create(prog='dot', format='svg'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(x_train, y_train, batch_size = 64, epochs = 25, shuffle=True, verbose=2, class_weight=class_weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"score = model.evaluate(x_test, y_test, verbose=0)\nprint(score)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}