{"cells":[{"metadata":{"_cell_guid":"d9d40dc8-eae2-5e9d-87cf-f15e58f82c29","_uuid":"0b4d7ae02f8e03f5faf2825cf2f1931901421e37","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nfrom matplotlib import pyplot as plt\nimport os\nfrom subprocess import check_output\nimport cv2\nfrom PIL import Image\nimport glob\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n#print(check_output([\"ls\",\"../input\"]).decode(\"utf8\"))\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"963016ae-a336-dc37-7d91-c41503710698","_uuid":"08333eef83f7f515fa852ce23121f599562e051b","trusted":true,"collapsed":true},"cell_type":"code","source":"trainLabels = pd.read_csv(\"../input/diabetic-retinopathy-detection/trainLabels.csv\")\ntrainLabels.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"5e72b0dc-412a-da1b-3b75-56811c08b05e","_uuid":"1dea0a128e317f36030310d4552934762211b7fa","trusted":true,"collapsed":true},"cell_type":"code","source":"filelist = glob.glob('../input/diabetic-retinopathy-detection/*.jpeg') \nnp.size(filelist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"226751b791f82f4899c6373df02c0ce04e9120b5","collapsed":true},"cell_type":"code","source":"# Load, resize and save the image data\nimg_data = []\nimg_label = []\nimg_r = 224\nimg_c = 224\nfor file in filelist:\n    tmp = cv2.imread(file)\n    tmp = cv2.resize(tmp,(img_r, img_c), interpolation = cv2.INTER_CUBIC)\n    tmp = cv2.cvtColor(tmp, cv2.COLOR_BGR2GRAY)\n    tmp = cv2.normalize(tmp, None, alpha=0, beta=1, norm_type=cv2.NORM_MINMAX, dtype=cv2.CV_32F)\n    img_data.append(np.array(tmp).flatten())\n    tmpfn = file\n    tmpfn = tmpfn.replace(\"../input/diabetic-retinopathy-detection/\",\"\")\n    tmpfn = tmpfn.replace(\".jpeg\",\"\")\n    img_label.append(trainLabels.loc[trainLabels.image==tmpfn, 'level'].values[0])\n#import pickle\n#with open('../diabetic-retinopathy-detection/img_data', 'wb') as f:\n#    pickle.dump(img_data, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"039cc87f0cc57cfae7081e8f64ba513319842b4b","collapsed":true},"cell_type":"code","source":"print(len(img_data))\nprint(len(img_label))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3707556541b4519121b9e346ee28f34eb3c5a4fc","collapsed":true},"cell_type":"code","source":"fileName = []\neye = []\nfor file in filelist:\n    tmpfn = file\n    tmpfn = tmpfn.replace(\"../input/diabetic-retinopathy-detection/\",\"\")\n    tmpfn = tmpfn.replace(\".jpeg\",\"\")\n    fileName.append(tmpfn)\n    if \"left\" in tmpfn:\n        eye.append(1)\n    else:\n        eye.append(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00fecd31c3ff161a403a198c57554828b37b71c0","collapsed":true},"cell_type":"code","source":"#data = pd.DataFrame({'fileName':fileName,'eye':eye,'img_data':img_data,'label':img_label}) # keyerror 10\ndata = pd.DataFrame({'eye':eye,'img_data':img_data,'label':img_label})\ndata.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c4aa7f16cbc576d2413413f01370e5e6ed6eb0e2","collapsed":true},"cell_type":"code","source":"data[['eye','label']].hist(figsize = (10, 5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"4f629fabf9b00efe749eafe617e09f0e105fcd18"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = data['img_data']\ny = data['label']\n#X = np.asarray(img_data)\n#y = np.asarray(img_label)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a13c4462f53fa164d6118da2dc82c2c0643ce20f","collapsed":true},"cell_type":"code","source":"from sklearn.utils import shuffle\n\ndata,label = shuffle(X_train,y_train, random_state=2)\ntrain_data = pd.DataFrame({'data': data, 'label':label})\ntrain_df = train_data.groupby(['label']).apply(lambda x: x.sample(160, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', train_df.shape[0], 'Old Size:', train_data.shape[0])\ntrain_df[['label']].hist(figsize = (10, 5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffd0f8ab535313cd13b9ccded480acfc8183062b","collapsed":true},"cell_type":"code","source":"#train_df = data\n#train_df[['label', 'eye']].hist(figsize = (10, 5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a72b84acbe05089bebec8aa7f052439afa8278f6","collapsed":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train = train_df['data']\ny_train = train_df['label']\n#X = np.asarray(img_data)\n#y = np.asarray(img_label)\n#X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9b506caa5868238682a7ea023abd67c77a579b2","collapsed":true},"cell_type":"code","source":"X_train = np.asarray(X_train)\ny_train = np.asarray(y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e005e9d1714e6409bc9bcb81f720b480a1eb854","collapsed":true},"cell_type":"code","source":"print(X_train.shape)\nprint(y_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb689a385a16700fc68309cb00d6622b86ca87cc","collapsed":true},"cell_type":"code","source":"# Reshaping the Training data for model input\n#print(type(X_train)) # <class 'pandas.core.series.Series'>\n#print(X_train.shape) # (800,)\nX_train_resh = np.zeros([X_train.shape[0],img_r, img_c, 1])\nfor i in range (X_train.shape[0]-1):\n    X_train_resh[i] = np.reshape(X_train[i], (img_r, img_c, 1))\n    #print(X_train_resh.shape) # (800,img_r,img_c,3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"2f8585b83999d37e2be048cfc512ffc1664c1be0"},"cell_type":"code","source":"X_test = np.asarray(X_test)\ny_test = np.asarray(y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06e6772512ba6e54180d02e51cf39474991b8c51","collapsed":true},"cell_type":"code","source":"X_test_resh = np.zeros([X_test.shape[0],img_r, img_c, 1])\nfor i in range (X_test.shape[0]-1):\n    X_test_resh[i] = np.reshape(X_test[i], (img_r, img_c, 1))\nprint(X_test_resh.shape) # (800,img_r,img_c,3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8787eae92878af72e2cd7de0f01b4320db4370b3","collapsed":true},"cell_type":"code","source":"from keras.utils import np_utils\n# convert class vectors to binary class matrices\nnb_classes = 5\nY_train = np_utils.to_categorical(y_train, nb_classes)\nY_test = np_utils.to_categorical(y_test, nb_classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"735945270a0624b974669180daf6c5e66944ef08","collapsed":true},"cell_type":"code","source":"#from sklearn.utils import shuffle\n#from scipy.sparse import coo_matrix\n#X_sparse = coo_matrix(X_train_resh)\n#X_train_resh, X_sparse, Y_train = shuffle(X_train_resh, X_sparse, Y_train, random_state=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c79e0502e5a7ec57e3dff2bdf613c1c790c3bd9c","collapsed":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib\n#plt.imshow(X_train_resh[100])\n\nimg=X_train_resh[100].reshape(img_r,img_c)\nplt.imshow(img)\nplt.imshow(img,cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3536ed74888017c33fabb9b8d5bfb49a3f03000","collapsed":true},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers.core import Dense, Dropout, Activation, Flatten\nfrom keras.layers.convolutional import Convolution2D, MaxPooling2D\nfrom keras.optimizers import SGD,RMSprop,adam\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras.optimizers import SGD\nfrom keras.layers import Dense, Dropout, Flatten\n# Create CNN model\nmodel = Sequential()\nmodel.add(Conv2D(64, (3, 3), activation='relu', input_shape = (img_r,img_c,1)))\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Conv2D(256, (3, 3), activation='relu'))\nmodel.add(Dropout(0.75))\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dense(5, activation='softmax'))\n\n# Calculate the class weights for unbalanced data\nfrom sklearn.utils.class_weight import compute_class_weight\nclasses = np.unique(y_train)\nclass_weight = compute_class_weight(\"balanced\", classes, y_train)\n\nsgd = SGD(lr=0.01, decay=1e-6, momentum=0.9, nesterov=True)\n# Compile model\nmodel.compile(loss='categorical_crossentropy', optimizer='sgd', metrics=['accuracy'])\n# convert class vectors to binary class matrices\nnb_classes = 5\nY_train = np_utils.to_categorical(y_train, nb_classes)\nY_test = np_utils.to_categorical(y_test, nb_classes)\n# Fit the model\nmodel.fit(X_train_resh, Y_train, batch_size = 32, epochs=30, verbose=1,class_weight=class_weight)\nscore = model.evaluate(X_test_resh, Y_test, verbose=0)\nprint(\"%s: %.2f%%\" % (model.metrics_names[1], score[1]*100))\n\nfrom keras.models import model_from_json\n# Model to JSON\nmodel_json = model.to_json()\nwith open(\"model_project_work.json\", \"w\") as json_file:\n    json_file.write(model_json)\n# Weights to HDF5\nmodel.save(\"model_project_work.h5\")\nprint(\"Saved model to disk\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2c009e8b7854171f43de54e9fba48b61d4d85c1","collapsed":true},"cell_type":"code","source":"# Load saved model\nfrom keras.models import load_model\nfrom keras.models import Sequential\nmodel = load_model('model_project_work.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d07e21ab83816369c4d774c46792dbefb25c190","collapsed":true},"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\npred_Y = model.predict(X_test_resh, batch_size = 32, verbose = True)\npred_Y_cat = np.argmax(pred_Y, -1)\ntest_Y_cat = np.argmax(Y_test, -1)\nprint('Accuracy on Test Data: %2.2f%%' % (accuracy_score(test_Y_cat, pred_Y_cat)))\nprint(classification_report(test_Y_cat, pred_Y_cat))\n\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\nsns.heatmap(confusion_matrix(test_Y_cat, pred_Y_cat), \n            annot=True, fmt=\"d\", cbar = False, cmap = plt.cm.Blues, vmax = X_test_resh.shape[0]//16)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"766b6c89135022fe69224bba3cde1d234991dc3e","collapsed":true},"cell_type":"code","source":"from sklearn.metrics import roc_curve, roc_auc_score\nsick_vec = test_Y_cat>0\nsick_score = np.sum(pred_Y[:,1:],1)\nfpr, tpr, _ = roc_curve(sick_vec, sick_score)\nfig, ax1 = plt.subplots(1,1, figsize = (6, 6), dpi = 150)\nax1.plot(fpr, tpr, 'b.-', label = 'Model Prediction (AUC: %2.2f)' % roc_auc_score(sick_vec, sick_score))\nax1.plot(fpr, fpr, 'g-', label = 'Random Guessing')\nax1.legend()\nax1.set_xlabel('False Positive Rate')\nax1.set_ylabel('True Positive Rate');","execution_count":null,"outputs":[]}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}