{"metadata":{"kernelspec":{"display_name":"Python 3 (ipykernel)","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.0"},"kaggle":{"accelerator":"none","dataSources":[],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy  as np \nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.cluster import KMeans\nfrom scipy import ndimage\nfrom skimage.filters import median\nfrom skimage.morphology import cube, disk\nimport seaborn as sns\nimport scipy\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import confusion_matrix, classification_report, ConfusionMatrixDisplay\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import GridSearchCV\n# import pywt\nimport pickle\nfrom sklearn import svm\nfrom sklearn.decomposition import PCA\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn import svm\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.naive_bayes import GaussianNB, BernoulliNB","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"aptos2019-blindness-detection/train.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['diagnosis'].value_counts()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n            # print(img.shape)\n        return img\ndef circle_crop(img, sigmaX=10):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    \n    img = crop_image_from_gray(img)    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width, depth = img.shape    \n    \n    x = int(width/2)-2\n    y = int(height/2)-2\n    r = np.amin((x,y))-3\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    return img ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def circle_edge_crop(img, sigmaX=10):   \n    \"\"\"\n    Create circular crop around image centre    \n    \"\"\"    \n    \n    img = crop_image_from_gray(img)    \n    # img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    height, width = img.shape    \n    \n    x = int(width/2)\n    y = int(height/2)\n    r = np.amin((x,y))\n    \n    circle_img = np.zeros((height, width), np.uint8)\n    cv2.circle(circle_img, (x,y), int(r), 1, thickness=-1)\n    img = cv2.bitwise_and(img, img, mask=circle_img)\n    img = crop_image_from_gray(img)\n    # img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    return img ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DR Affected","metadata":{}},{"cell_type":"code","source":"img = cv2.imread(f\"aptos2019-blindness-detection/train_images/{df['id_code'][5]}.png\")\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nplt.imshow(img)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gray image\nimg1 = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\nprint(img1.shape)\nplt.imshow(img1, cmap='gray')\nplt.show()\nimg2=circle_crop(img)  # image is cropped\nprint(img2.shape)\nplt.imshow(img2, cmap='gray')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### image is resized to 350 X 354","metadata":{}},{"cell_type":"code","source":"img2= cv2.resize(img2, (350, 354))\nplt.imshow(img2, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# performing discrete wavelet transform 'Haar' to the image for denoising and edge detection\ncoeffs = pywt.dwt2(img2, 'haar')\nimg_dwt = pywt.idwt2(coeffs, 'haar')\nplt.imshow(img_dwt, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createMatchedFilterBank():\n    filters = []\n    ksize = 20\n    for theta in np.arange(0, np.pi, np.pi/16):\n        kern = cv2.getGaborKernel((ksize, ksize), 6, theta, 12, 0.37, 0, ktype=cv2.CV_32F)\n        kern /= 1.5*kern.sum()\n        filters.append(kern)\n    return filters\ndef applyFilters(im, kernels):\n    images = np.array([cv2.filter2D(im, -1, k) for k in kernels])\n    return np.max(images, 0)\n\nbank_gf = createMatchedFilterBank()\nimm_gauss2 = []\nimg_new = applyFilters(img_dwt, bank_gf)\nplt.imshow(img_new, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_new.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### cropping the edge of the circle","metadata":{}},{"cell_type":"code","source":"img_edge_cropped_1 = circle_edge_crop(img_new)\nplt.imshow(img_edge_cropped_1, cmap='gray') ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_edge_cropped_1.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Image Segmentation","metadata":{}},{"cell_type":"code","source":"img_edge_cropped_1 = cv2.resize(img_edge_cropped_1, (350, 354))\nZ = img_edge_cropped_1.reshape((-1, 3))\nZ = np.float32(Z)\nk = cv2.KMEANS_PP_CENTERS\ncriteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 10, 1.0)\nK = 15\nret, label, center = cv2.kmeans(Z, K, None, criteria, 10, k)\ncentrer = np.uint8(center)\nsegmented_img_1 = center[label.flatten()].reshape(img_edge_cropped_1.shape)\nprint(segmented_img_1.shape)\nplt.imshow(segmented_img_1, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(segmented_img_1.reshape(-1)).describe()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normal Eye","metadata":{}},{"cell_type":"code","source":"img = cv2.imread(f\"aptos2019-blindness-detection/train_images/{df['id_code'][6]}.png\")\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nplt.imshow(img)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Cropping image","metadata":{}},{"cell_type":"code","source":"# gray image\nimg1 = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\nprint(img1.shape)\nplt.imshow(img1, cmap='gray')\nplt.show()\nimg2=circle_crop(img)  # image is cropped\nplt.imshow(img2, cmap='gray')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Image Resizing","metadata":{}},{"cell_type":"code","source":"img2 = cv2.resize(img2, (350, 354))\nplt.imshow(img2, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Performing Discrete Wavelet Trasnform 'Haar'","metadata":{}},{"cell_type":"code","source":"coeffs = pywt.dwt2(img2, 'haar')\nimg_dwt = pywt.idwt2(coeffs, 'haar')\nplt.imshow(img_dwt, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Edge Detection","metadata":{}},{"cell_type":"code","source":"def createMatchedFilterBank():\n    filters = []\n    ksize = 20\n    for theta in np.arange(0, np.pi, np.pi/16):\n        kern = cv2.getGaborKernel((ksize, ksize), 6, theta, 12, 0.37, 0, ktype=cv2.CV_32F)\n        kern /= 1.5*kern.sum()\n        filters.append(kern)\n    return filters\ndef applyFilters(im, kernels):\n    images = np.array([cv2.filter2D(im, -1, k) for k in kernels])\n    return np.max(images, 0)\n\nbank_gf = createMatchedFilterBank()\nimm_gauss2 = []\nimg_new = applyFilters(img_dwt, bank_gf)\nplt.imshow(img_new, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### cropping the edge of the circle","metadata":{}},{"cell_type":"code","source":"img_edge_cropped_2 = circle_edge_crop(img_new)\nplt.imshow(img_edge_cropped_2, cmap='gray') ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Image Segmentation","metadata":{}},{"cell_type":"code","source":"img_edge_cropped_2 = cv2.resize(img_edge_cropped_2, (350, 354))\nZ = img_edge_cropped_2.reshape((-1, 3))\nZ = np.float32(Z)\nk = cv2.KMEANS_PP_CENTERS\ncriteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 10, 1.0)\nK = 15\nret, label, center = cv2.kmeans(Z, K, None, criteria, 10, k)\ncentrer = np.uint8(center)\nsegmented_img_2 = center[label.flatten()].reshape(img_edge_cropped_2.shape)\nprint(segmented_img_2.shape)\nplt.imshow(segmented_img_2, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(segmented_img_2.reshape(-1)).describe()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comparison between Diabetic Retinopathy and Normal Eye Image","metadata":{}},{"cell_type":"code","source":"# distribution of DR affected eye image and Normal eye image\naffected = cv2.imread(f\"aptos2019-blindness-detection/train_images/{df['id_code'][0]}.png\")\naffected = cv2.cvtColor(affected, cv2.COLOR_BGR2GRAY)\nnormal = cv2.imread(f\"aptos2019-blindness-detection/train_images/{df['id_code'][3]}.png\")\nnormal = cv2.cvtColor(normal, cv2.COLOR_BGR2GRAY)\nsns.kdeplot(affected.reshape(-1), label='DR Affected')\nsns.kdeplot(normal.reshape(-1), label='Normal')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### After Edge Detection","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(img_edge_cropped_1.reshape(-1), label='DR Affected')\nsns.kdeplot(img_edge_cropped_2.reshape(-1), label='Normal')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### After Segmentation","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(segmented_img_1.reshape(-1), label='Segmented Image of DR Affected Eye')\nsns.kdeplot(segmented_img_2.reshape(-1), label='Segmented Image of Normal Eye')\nplt.legend(fontsize=8)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Converting 5 class into 2 class","metadata":{}},{"cell_type":"code","source":"for i in range(df.shape[0]):\n    if df['diagnosis'][i]!=0:\n        df.iloc[i, 1] = 1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['diagnosis'].value_counts()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['diagnosis'].value_counts().plot(kind='bar')\nplt.title(\" Value Counts of DR eye and Normal eye\", size=10, loc='center')\nplt.xlabel('   DR                                                 Normal')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Making Dataset","metadata":{}},{"cell_type":"code","source":"def segmentation_func(img_edge_cropped):\n    img_edge_cropped = cv2.resize(img_edge_cropped, (350, 354))\n    Z = img_edge_cropped.reshape((-1, 3))\n    Z = np.float32(Z)\n    k = cv2.KMEANS_PP_CENTERS\n    criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 10, 1.0)\n    K = 15\n    ret, label, center = cv2.kmeans(Z, K, None, criteria, 10, k)\n    centrer = np.uint8(center)\n    segmented_img = center[label.flatten()].reshape(img_edge_cropped.shape)\n    return segmented_img","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"statistical_dataset = []\nimage_dataset = []\nbank_gf = createMatchedFilterBank()\nfor i in tqdm(range(0, df.shape[0])):\n    img = cv2.imread(f\"aptos2019-blindness-detection/train_images/{df['id_code'][i]}.png\") # read image\n    cropped_img=circle_crop(img) # cropping \n    cropped_img = cv2.resize(cropped_img, (350, 354)) # rizing\n    coeffs = pywt.dwt2(cropped_img, 'haar')\n    img_dwt = pywt.idwt2(coeffs, 'haar')\n    img_new = applyFilters(img_dwt, bank_gf)\n    img_edge_cropped = circle_edge_crop(img_new)\n    segmented_img = segmentation_func(img_edge_cropped)\n    segmented_img_1d = segmented_img.reshape(-1)\n    image_dataset.append(segmented_img_1d)\n    \n    des = pd.DataFrame(segmented_img_1d).describe()\n    std = des.iloc[2][0]\n    mean = des.iloc[1][0]\n    min = des.iloc[3][0]\n    max = des.iloc[7][0]\n    mdian = des.iloc[5][0]\n    lower_quartile = des.iloc[4][0]\n    upper_quartile = des.iloc[6][0]\n    skewness = scipy.stats.skew(segmented_img_1d)\n    mad = np.mean(np.absolute(segmented_img_1d - np.mean(segmented_img_1d)))  # mean absolute deviation\n    statistical_dataset.append([std, mean, min, max, mdian, lower_quartile, upper_quartile, skewness, mad, df['diagnosis'][i]])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(image_dataset).shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(statistical_dataset).shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"image_dataset.pkl\", 'wb') as f:\n    pickle.dump(image_dataset, f)\nwith open(\"statistical_dataset.pkl\", 'wb') as f:\n    pickle.dump(statistical_dataset, f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"image_dataset.pkl\", 'rb') as f:\n    image_dataset = pickle.load(f)\nwith open(\"statistical_dataset.pkl\", 'rb') as f:\n    statistical_dataset = pickle.load(f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"statistical_df = pd.DataFrame(statistical_dataset)\nstatistical_df.columns = ['std', 'mean', 'min', 'max', 'median', 'lower_quartile', 'upper_quartile', 'skewness', 'mad', 'label']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"statistical_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Scaling","metadata":{}},{"cell_type":"markdown","source":"#### For statistical Features","metadata":{}},{"cell_type":"code","source":"X_1 = statistical_df.iloc[:, :-1] \ny_1 = statistical_df.iloc[:, -1]\nX_train_1, X_test_1, y_train_1, y_test_1 = train_test_split(X_1, y_1, test_size=0.2)\nscaler = StandardScaler()\nX_train_1 = scaler.fit_transform(X_train_1)\nX_test_1 = scaler.transform(X_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### For Image","metadata":{}},{"cell_type":"code","source":"X_2 = image_dataset[:, :-1] \ny_2 = image_dataset[:, -1]\nX_train_2, X_test_2, y_train_2, y_test_2 = train_test_split(X_2, y_2, test_size=0.2)\nX_train_2 = X_train_2/255\nX_test_2 = X_test_2/255","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SVM - using statistical features","metadata":{}},{"cell_type":"code","source":"from sklearn import svm\nclf = svm.SVC(C=1000, gamma=0.01)\nclf.fit(X_train_1, y_train_1)\ny_pred = clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\n# target_names = ['good 0', 'medium 1', 'bad 2']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# grid search for finding best hyper parameters\nparam_grid =  {'C': [0.1, 1, 10, 100, 1000], 'gamma': [1, 0.1, 0.01, 0.001, 0.0001], 'kernel':['rbf']}\nclf = svm.SVC()\ngrid_search = GridSearchCV(clf, param_grid, cv=5,\n scoring='accuracy',\nreturn_train_score=True)\ngrid_search.fit(X_train_1, y_train_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.best_estimator_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### Cross Val accuracy using statistical features","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nX_1_s = scaler.fit_transform(X_1)\nnp.mean(cross_val_score(svm.SVC(C=100, gamma=0.01), X_1_s, y_1, scoring=\"accuracy\", cv=10))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dataset = np.array(image_dataset)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SVM - using image","metadata":{}},{"cell_type":"code","source":"from sklearn import svm\nclf = svm.SVC()\nclf.fit(X_train_2, y_train_2)\ny_pred = clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\n# target_names = ['good 0', 'medium 1', 'bad 2']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### cross val score","metadata":{}},{"cell_type":"code","source":"X_2_s = X_2/255\nnp.mean(cross_val_score(svm.SVC(), X_2_s, y_2, scoring=\"accuracy\", cv=5))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Classifier - using Statistical Data","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nforest_clf = RandomForestClassifier(criterion='entropy', max_features=2, n_estimators=70)\nforest_clf.fit(X_train_1, y_train_1)\ny_pred = forest_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\n# target_names = ['good 0', 'medium 1', 'bad 2']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid =  {'n_estimators': [2, 3, 5, 10, 20, 30, 50, 70,  100], 'criterion': ['gini', 'entropy', 'log_loss'], 'max_features':[2, 3, 4, 5, 6]}\nclf = RandomForestClassifier()\ngrid_search = GridSearchCV(clf, param_grid, cv=5,\n scoring='accuracy',\nreturn_train_score=True)\ngrid_search.fit(X_train_1, y_train_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.best_estimator_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cross Val Score\nscaler = StandardScaler()\nX_1_s = scaler.fit_transform(X_1)\nnp.mean(cross_val_score(RandomForestClassifier(criterion='entropy', max_features=2, n_estimators=70), X_1_s, y_1, scoring=\"accuracy\", cv=10))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Classifier - using Image Data","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nforest_clf = RandomForestClassifier()\nforest_clf.fit(X_train_2, y_train_2)\ny_pred = forest_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\n# target_names = ['good 0', 'medium 1', 'bad 2']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression - using Statistical Data","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlog_clf = LogisticRegression()\nlog_clf.fit(X_train_1, y_train_1)\ny_pred = log_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression - using image Data","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlog_clf = LogisticRegression(max_iter=200)\nlog_clf.fit(X_train_2, y_train_2)\ny_pred = log_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\n# target_names = ['good 0', 'medium 1', 'bad 2']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decision Tree Classifier - using statistical data","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = DecisionTreeClassifier(criterion='gini', splitter='best', max_depth=None, min_samples_split=5, min_samples_leaf=1,\n                             max_features=None, max_leaf_nodes=None, min_impurity_decrease=0.01)\nmodel.fit(X_train_1, y_train_1)\ny_pred = model.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decision Tree Classifier - using Image data","metadata":{}},{"cell_type":"code","source":"model = DecisionTreeClassifier()\nmodel.fit(X_train_2, y_train_2)\ny_pred = model.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN - using statistical data","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_clf = KNeighborsClassifier(n_neighbors=5, weights='uniform')\nmodel_clf.fit(X_train_1, y_train_1)\ny_pred = model_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN - using image data","metadata":{}},{"cell_type":"code","source":"model_clf = KNeighborsClassifier(n_neighbors=5, weights='distance')\nmodel_clf.fit(X_train_2, y_train_2)\ny_pred = model_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost - using Statistical data","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_clf = XGBClassifier(learning_rate=0.1, n_estimators=120)\nmodel_clf.fit(X_train_1, y_train_1)\ny_pred = model_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nX_1_s = scaler.fit_transform(X_1)\nnp.mean(cross_val_score(XGBClassifier(learning_rate=0.1, n_estimators=120), X_1_s, y_1, scoring=\"accuracy\", cv=10))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ADABoost - using Statistical data","metadata":{}},{"cell_type":"code","source":"model_clf = AdaBoostClassifier(algorithm='SAMME', n_estimators=100, learning_rate=0.2)\nmodel_clf.fit(X_train_1, y_train_1)\ny_pred = model_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid =  {'n_estimators': [2, 3, 5, 10, 20, 30, 50, 70,  100], 'algorithm': ['SAMME'], 'learning_rate':[0.1, 0.01, 0.09, 0.2, 0.05]}\n\nclf = AdaBoostClassifier()\ngrid_search = GridSearchCV(clf, param_grid, cv=5,\n scoring='accuracy',\nreturn_train_score=True)\ngrid_search.fit(X_train_1, y_train_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search.best_estimator_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ADABoost - using image data","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_clf = AdaBoostClassifier(algorithm='SAMME', n_estimators=100, learning_rate=0.2)\nmodel_clf.fit(X_train_2, y_train_2)\ny_pred = model_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Ensemble model","metadata":{}},{"cell_type":"markdown","source":"###### There are four most commonly used ensemble models.\n 1. Voting\n 2. Bagging\n 3. Boosting (ex - ADABoost)\n 4. Stacking ","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### (RF + SVM + LogisticRegression) - using statistical data","metadata":{}},{"cell_type":"code","source":"voting_clf = VotingClassifier(\n estimators=[('rf', RandomForestClassifier(criterion='entropy', max_features=2, n_estimators=70)), ('svc', svm.SVC(C=100, gamma=0.01)), ('lor', LogisticRegression(max_iter=200))],\n voting='hard')\nvoting_clf.fit(X_train_1, y_train_1)\ny_pred = voting_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### (RF + SVM + LogisticRegression) - using image data","metadata":{}},{"cell_type":"code","source":"voting_clf = VotingClassifier(\n estimators=[('rf', RandomForestClassifier(criterion='entropy', max_features=2, n_estimators=70)), ('svc', svm.SVC()), ('lor', LogisticRegression(max_iter=300))],\n voting='hard')\nvoting_clf.fit(X_train_2, y_train_2)\ny_pred = voting_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Stacking ","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import StackingClassifier","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### using statistical data","metadata":{}},{"cell_type":"code","source":"rfc = RandomForestClassifier()\nknn = KNeighborsClassifier()\nSVM = svm.SVC()\nlr = LogisticRegression(max_iter=300)\nstack_model = StackingClassifier(estimators=[('rf', rfc), ('svc', SVM), ('knn', knn)], final_estimator=lr)\nstack_model.fit(X_train_1, y_train_1)\ny_pred = stack_model.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### using image data","metadata":{}},{"cell_type":"code","source":"rfc = RandomForestClassifier()\nknn = KNeighborsClassifier()\nSVM = svm.SVC()\nlr = LogisticRegression(max_iter=300)\nstack_model = StackingClassifier(estimators=[('rf', rfc), ('svc', SVM), ('knn', knn)], final_estimator=lr)\nstack_model.fit(X_train_2, y_train_2)\ny_pred = stack_model.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bagging","metadata":{}},{"cell_type":"markdown","source":"#### SVM","metadata":{}},{"cell_type":"markdown","source":"###### for statistical data","metadata":{}},{"cell_type":"code","source":"bag_clf = BaggingClassifier(\n svm.SVC())\nbag_clf.fit(X_train_1, y_train_1)\ny_pred = bag_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### for image data","metadata":{}},{"cell_type":"code","source":"bag_clf = BaggingClassifier(\n svm.SVC())\nbag_clf.fit(X_train_2, y_train_2)\ny_pred = bag_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### RandomForestClassifier","metadata":{}},{"cell_type":"markdown","source":"#### Using statistical data","metadata":{}},{"cell_type":"code","source":"bag_clf = BaggingClassifier(\n  RandomForestClassifier())\nbag_clf.fit(X_train_1, y_train_1)\ny_pred = bag_clf.predict(X_test_1)\naccuracy_score(y_pred, y_test_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_1, y_pred, target_names=target_names))\nconfusionmatrix = confusion_matrix(y_test_1, y_pred)\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### using image","metadata":{}},{"cell_type":"code","source":"bag_clf = BaggingClassifier(\nRandomForestClassifier())\nbag_clf.fit(X_train_2, y_train_2)\ny_pred = bag_clf.predict(X_test_2)\naccuracy_score(y_pred, y_test_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_names = ['Normal 0', 'Affected 1']\nprint(classification_report(y_test_2, y_pred, target_names=target_names))\n\nconfusionmatrix = confusion_matrix(y_test_2, y_pred)\n\ncm_display = ConfusionMatrixDisplay(confusion_matrix = confusionmatrix)\ncm_display.plot()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]}]}