{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import json\nimport math\nimport os\n\nimport cv2\nfrom PIL import Image\nimport numpy as np\nimport seaborn as sns\nfrom keras import layers\nfrom keras.applications import DenseNet121, MobileNetV2\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential, load_model\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, confusion_matrix\nimport scipy\nimport tensorflow as tf\nfrom tqdm import tqdm\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Set random seed for reproducibility."},{"metadata":{},"cell_type":"markdown","source":"# Loading & Exploration"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/valid-and-test-ta/x_train_8.csv')\nvalid_df = pd.read_csv('../input/valid-and-test-ta/x_valid_8.csv')\ntest_df = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint(train_df.shape)\nprint(valid_df.shape)\nprint(test_df.shape)\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#resample\nfrom sklearn.utils import resample\nX=train_df\nnormal=X[X.diagnosis==0]\nmild=X[X.diagnosis==1]\nmoderate=X[X.diagnosis==2]\nsevere=X[X.diagnosis==3]\npdr=X[X.diagnosis==4]\n\n#downsampled\nmild = resample(mild,\n                replace=True, # sample with replacement\n                n_samples=700, # match number in majority class\n                random_state=2020) # reproducible results\nmoderate = resample(moderate,\n                    replace=False, # sample with replacement\n                    n_samples=700, # match number in majority class\n                    random_state=2020) # reproducible results\nsevere = resample(severe,\n                  replace=True, # sample with replacement\n                  n_samples=700, # match number in majority class\n                  random_state=2020) # reproducible results\nnormal = resample(normal,\n                  replace=False, # sample with replacement\n                  n_samples=700, # match number in majority class\n                  random_state=2020) # reproducible results\npdr = resample(pdr,\n               replace=True, # sample with replacement\n               n_samples=700, # match number in majority class\n               random_state=2020) # reproducible results    \n\n# combine minority and downsampled majority\nsampled = pd.concat([normal, mild, moderate, severe, pdr])\n\n# checking counts\nsampled.diagnosis.value_counts()\n\ntrain_df = sampled\ntrain_df = train_df.sample(frac=1).reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Mengecek apakah ukuran sudah sesuai\nprint('Number of train samples: ', train_df.shape[0])\nprint('Number of test samples: ', valid_df.shape[0])\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_df['diagnosis'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Resize Images\n\nWe will resize the images to 224x224, then create a single numpy array to hold the data."},{"metadata":{"trusted":true},"cell_type":"code","source":"def preprocess_image(image_path, desired_size=224):\n    im = Image.open(image_path)\n    im = im.resize((desired_size, )*2, resample=Image.BILINEAR)\n    \n    return im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"N = train_df.shape[0]\nx_train = np.empty((N, 224, 224, 3), dtype=np.float32)\n\nfor i, image_id in enumerate(tqdm(train_df['id_code'])):\n    x_train[i, :, :, :] = preprocess_image(\n        f'../input/aptos2019-blindness-detection/train_images/{image_id}.png'\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"N = valid_df.shape[0]\nx_val = np.empty((N, 224, 224, 3), dtype=np.float32)\n\nfor i, image_id in enumerate(tqdm(valid_df['id_code'])):\n    x_val[i, :, :, :] = preprocess_image(\n        f'../input/aptos2019-blindness-detection/train_images/{image_id}.png'\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"N = test_df.shape[0]\nx_test = np.empty((N, 224, 224, 3), dtype=np.float32)\n\nfor i, image_id in enumerate(tqdm(test_df['id_code'])):\n    x_test[i, :, :, :] = preprocess_image(\n        f'../input/aptos2019-blindness-detection/test_images/{image_id}.png'\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train = train_df['diagnosis']\ny_val = valid_df['diagnosis']\nprint(x_train.shape)\nprint(y_train.shape)\nprint(x_val.shape)\nprint(y_val.shape)\nprint(x_test.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Model: MobilenetV2"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = load_model('../input/my-best/model_8.h5')\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submit"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_pred = model.predict(x_train)\ny_train_pred = train_pred\ny_train_pred = np.clip(y_train_pred,0,4)\ny_train_pred = y_train_pred.astype(int)\n\nlabels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ncnf_matrix = confusion_matrix(train_df['diagnosis'].astype('int'), y_train_pred)\ndf_cm = pd.DataFrame(cnf_matrix, index=labels, columns=labels)\nplt.figure(figsize=(16, 7))\nsns.heatmap(df_cm, annot=True, cmap=\"Blues\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kappa_val = cohen_kappa_score(\n            train_df['diagnosis'].astype('int'),\n            y_train_pred, \n            weights='quadratic'\n        )\nkappa_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_pred = model.predict(x_val)\ny_val_pred = val_pred\ny_val_pred = np.clip(y_val_pred,0,4)\ny_val_pred = y_val_pred.astype(int)\n\nlabels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ncnf_matrix = confusion_matrix(valid_df['diagnosis'].astype('int'), y_val_pred)\ndf_cm = pd.DataFrame(cnf_matrix, index=labels, columns=labels)\nplt.figure(figsize=(16, 7))\nsns.heatmap(df_cm, annot=True, cmap=\"Blues\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kappa_val = cohen_kappa_score(\n            valid_df['diagnosis'].astype('int'),\n            y_val_pred, \n            weights='quadratic'\n        )\nkappa_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import classification_report\ntarget_names = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\nprint(classification_report(valid_df['diagnosis'].astype('int'), y_val_pred, target_names=target_names))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.DataFrame()\nvalid_data = pd.DataFrame()\ntrain_data['trainLabel'] = train_df['diagnosis']\ntrain_data['trainPred'] = train_pred\nvalid_data['validLabel'] = valid_df['diagnosis']\nvalid_data['validPred'] = val_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = [0,1,2,3,4]\n\n# Iterate through the five airlines\nfor label in labels:\n    # Subset to the airline\n    subset = train_data[train_data['trainLabel'] == label]\n    \n    # Draw the density plot\n    sns.distplot(subset['trainPred'], hist = False, kde = True,\n                 kde_kws = {'linewidth': 3},\n                 label = label)\n    \n# Plot formatting\nplt.legend(prop={'size': 10}, title = 'Kelas')\nplt.title('Density Plot with Multiple Classes')\nplt.xlabel('Prediksi')\nplt.ylabel('Density')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = [0,1,2,3,4]\n\n# Iterate through the five airlines\nfor label in labels:\n    # Subset to the airline\n    subset = valid_data[valid_data['validLabel'] == label]\n    \n    # Draw the density plot\n    sns.distplot(subset['validPred'], hist = False, kde = True,\n                 kde_kws = {'linewidth': 3},\n                 label = label)\n    \n# Plot formatting\nplt.legend(prop={'size': 10}, title = 'Kelas')\nplt.title('Density Plot with Multiple Classes')\nplt.xlabel('Prediksi')\nplt.ylabel('Density')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_pred = model.predict(x_test)\nfrom sklearn.naive_bayes import GaussianNB\ngnb = GaussianNB()\ny_test_pred = gnb.fit(val_pred, valid_data['validLabel']).predict(test_pred)\n\ntest_df['diagnosis'] = y_test_pred\ntest_df.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\npred = model.predict(x_test)\ny_val_pred = pred\ny_val_pred = np.clip(y_val_pred,0,4)\ny_val_pred = y_val_pred.astype(int)\n\ntest_df['diagnosis'] = y_val_pred\ntest_df.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras\nlayer_name = 'dense_2'\nintermediate_layer_model = keras.Model(inputs=model.input,\n                                       outputs=model.get_layer(layer_name).output)\nintermediate_layer_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train_pred = intermediate_layer_model.predict(x_train)\ny_valid_pred = intermediate_layer_model.predict(x_val)\ny_test_pred = intermediate_layer_model.predict(x_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import label_binarize\nfrom scipy import interp\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.multiclass import OneVsOneClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_val = label_binarize(y_val, classes=[0,1,2,3,4])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_classes = 5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# classifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nclf = make_pipeline(StandardScaler(),SVC(probability=True))\nclf = OneVsOneClassifier(clf)\ny_score = clf.fit(y_train_pred, y_train).decision_function(y_valid_pred)\n\nkappa_val = cohen_kappa_score(\n            np.argmax(y_val,axis=1),\n            np.argmax(y_score,axis=1), \n            weights='quadratic'\n        )\n\nprint(kappa_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Compute ROC curve and ROC area for each class\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(n_classes):\n    fpr[i], tpr[i], _ = roc_curve(y_val[:, i], y_score[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Compute micro-average ROC curve and ROC area\nfpr[\"micro\"], tpr[\"micro\"], _ = roc_curve(y_val.ravel(), y_score.ravel())\nroc_auc[\"micro\"] = auc(fpr[\"micro\"], tpr[\"micro\"])\n\nfrom itertools import cycle\nlw = 2\nall_fpr = np.unique(np.concatenate([fpr[i] for i in range(n_classes)]))\n\n# Then interpolate all ROC curves at this points\nmean_tpr = np.zeros_like(all_fpr)\nfor i in range(n_classes):\n    mean_tpr += interp(all_fpr, fpr[i], tpr[i])\n\n# Finally average it and compute AUC\nmean_tpr /= n_classes\n\nfpr[\"macro\"] = all_fpr\ntpr[\"macro\"] = mean_tpr\nroc_auc[\"macro\"] = auc(fpr[\"macro\"], tpr[\"macro\"])\n\n# Plot all ROC curves\nplt.figure()\nplt.plot(fpr[\"macro\"], tpr[\"macro\"],\n         label='macro-average ROC curve (area = {0:0.2f})'\n               ''.format(roc_auc[\"macro\"]),\n         color='navy', linestyle=':', linewidth=4)\n\ncolors = cycle(['aqua', 'darkorange', 'cornflowerblue'])\nfor i, color in zip(range(n_classes), colors):\n    plt.plot(fpr[i], tpr[i], color=color, lw=lw,\n             label='ROC curve of class {0} (area = {1:0.2f})'\n             ''.format(i, roc_auc[i]))\n\nplt.plot([0, 1], [0, 1], 'k--', lw=lw)\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Some extension of Receiver operating characteristic to multi-class')\nplt.legend(loc=\"lower right\")\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\ncnf_matrix = confusion_matrix(np.argmax(y_val,axis=1), np.argmax(y_score,axis=1))\ndf_cm = pd.DataFrame(cnf_matrix, index=labels, columns=labels)\nplt.figure(figsize=(16, 7))\nsns.heatmap(df_cm, annot=True, cmap=\"Blues\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import classification_report\ntarget_names = ['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative DR']\nprint(classification_report(np.argmax(y_val,axis=1), np.argmax(y_score,axis=1), target_names=target_names))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\ny_score = clf.decision_function(y_test_pred)\ntest_df['diagnosis'] = np.argmax(y_score,axis=1)\ntest_df.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}