{"cells":[{"metadata":{},"cell_type":"markdown","source":"# <div style=' background-color: #ddffff; border-left: 10px solid #2196F3; padding: 20px;'> Playing with the data! </div>"},{"metadata":{},"cell_type":"markdown","source":"### <div style=' color: white; background-color: #2D93D5; border-left: 10px solid #014F99; padding: 20px;'> Simplified Problem Statement </div>\n- The problem is -\n    - From a labelled dataset of 3662 retina images. Teach / Train a Machine Learning model to correctly identify the severity of diabetic retinopathy on a scale of 0 to 4.\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import re\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn import tree\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nimport lightgbm as lgb\nimport xgboost as xg\n\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import KFold","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### <div style=' color: white; background-color: #2D93D5; border-left: 10px solid #014F99; padding: 20px;'> Data Description </div>\n- **Total entries in training set - 3662**\n    - *0 - No DR - 1805 (49.29%)*\n    - *1 - Mild - 999 (27.28%)*\n    - *2 - Moderate - 370 (10.10%)*\n    - *3 - Severe - 295 (8.05%)*\n    - *4 - Proliferative DR - 193 (5.27%)*"},{"metadata":{},"cell_type":"markdown","source":"Note - All files are of format .png which are being resized to (512 x 512) dimensions and being read as grayscale images."},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#Prepping X_train\n#Note - 512 x 512 - 7.2 GB\n\nImg_size = 512  \nX0 = []\nbase_pth = '../input/train_images/'\n\nfor filename in os.listdir('../input/train_images/'):\n    ds = cv2.imread(str(base_pth+filename),0)\n    b0 = cv2.resize(ds,(Img_size,Img_size))\n    X0.append(b0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Prepping Y_train\n\ndf_train = pd.read_csv('../input/train.csv')\ny = []\n\nfor filename in os.listdir('../input/train_images/'):\n    idz = str(filename)[:-4]\n    for idx,each in enumerate(df_train['id_code']):\n        if idz == each:\n            y.append(int(df_train.iloc[idx,1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = np.array([i[0] for i in X0]).reshape((-1,512))\ny = np.array(y)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### <div style=' color: white; background-color: #2D93D5; border-left: 10px solid #014F99; padding: 20px;'>Survey</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"def conf_matrix(y_pred,y_actual):\n        \n    cm = confusion_matrix(y_actual, y_pred)\n    acc = (y_actual == y_pred).sum()/len(y_actual)\n\n    return cm,acc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def print_metrics(cm,acc):\n    print(\"Confusion Matrix - \")\n    print(cm)\n    print(\"Accuracy\")\n    print(acc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_test_func(algo_init_model):\n    \n    #Add k_Fold step here\n    k = 5\n    kf = KFold(n_splits=k)\n    \n    avg_cm = np.zeros((5,5))\n    avg_acc = 0\n    \n    for train_index, test_index in kf.split(X):\n        X_train, X_test = X[train_index], X[test_index]\n        y_train, y_test = y[train_index], y[test_index]\n        \n        #Training\n        y_pred = algo_init_model.fit(X_train, y_train)\n\n        #Prediction\n        y_pred_val = y_pred.predict(X_test)\n\n        #Metrics\n        cm, acc = conf_matrix(y_pred_val,y_test)\n        avg_acc += acc\n        avg_cm += cm\n\n    print_metrics(avg_cm/k,avg_acc/k)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>Multi-Nomial Naive Bayes</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on Naive Bayes\ngnb = MultinomialNB()\ntrain_test_func(gnb)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>Decision Tree</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on Decision Tree\nclf = tree.DecisionTreeClassifier()\ntrain_test_func(clf)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>Logistic Regression</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on Logistic Regression\nclf = LogisticRegression(random_state=0, solver='lbfgs', multi_class='multinomial')\ntrain_test_func(clf)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>Random Forest</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on Random Forest\nclf = RandomForestClassifier(n_estimators=100, max_depth=2,random_state=0)\ntrain_test_func(clf)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>AdaBoost</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on AdaBoost\nclf = AdaBoostClassifier(n_estimators=50,learning_rate=1,random_state=0)\ntrain_test_func(clf)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>KNN</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on KNN\nneigh = KNeighborsClassifier(n_neighbors=5)\ntrain_test_func(neigh)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>LightGBM</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on LightGBM\n\nparams = {}\nparams['learning_rate'] = 0.003\nparams['boosting_type'] = 'gbdt'\nparams['objective'] = 'multiclass'\nparams['metric'] = 'multi_logloss'\nparams['sub_feature'] = 0.5\nparams['num_leaves'] = 30\nparams['min_data'] = 50\nparams['max_depth'] = 20\nparams['num_class'] = 5\n\nk = 5\nkf = KFold(n_splits=k)\n    \navg_cm = np.zeros((5,5))\navg_acc = 0\n    \nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X[train_index], X[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n        \n    #Training\n    d_train = lgb.Dataset(X_train, label=y_train)\n    y_pred = lgb.train(params, d_train, 100)\n\n    #Prediction\n    y_pred_val0 = y_pred.predict(X_test)\n    print(y_pred_val0[0])\n    \n    y_pred_val = []\n\n    for x in y_pred_val0:\n        print(np.argmax(x))\n        y_pred_val.append(np.argmax(x))\n    \n    #Metrics\n    cm, acc = conf_matrix(y_pred_val,y_test)\n    avg_acc += acc\n    avg_cm += cm\nprint_metrics(avg_cm/k,avg_acc/k)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### <div style=' color: black; background-color: #C7D86F; border-left: 10px solid #F7C407; padding: 20px;'>XGBoost</div>"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Trying on XGBoost\nxgbt = xg.XGBClassifier()\ntrain_test_func(xgbt)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}