{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Standard dependencies\nimport cv2\nimport time\nimport scipy as sp\nimport numpy as np\nimport random as rn\nimport pandas as pd\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom functools import partial\nimport matplotlib.pyplot as plt\n\n# Machine Learning\nimport tensorflow as tf\nimport keras\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom sklearn.metrics import cohen_kappa_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-29T20:04:40.092924Z","iopub.execute_input":"2022-05-29T20:04:40.093895Z","iopub.status.idle":"2022-05-29T20:04:40.099585Z","shell.execute_reply.started":"2022-05-29T20:04:40.093854Z","shell.execute_reply":"2022-05-29T20:04:40.098918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install efficientnet","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:40.133535Z","iopub.execute_input":"2022-05-29T20:04:40.134415Z","iopub.status.idle":"2022-05-29T20:04:50.275553Z","shell.execute_reply.started":"2022-05-29T20:04:40.13436Z","shell.execute_reply":"2022-05-29T20:04:50.274646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"#Define path of files\n\nKAGGLE_DIR = '../input/aptos2019-blindness-detection/'\nTRAIN_DF_PATH = KAGGLE_DIR + \"train.csv\"\nTEST_DF_PATH = KAGGLE_DIR + 'test.csv'\nTRAIN_IMG_PATH = KAGGLE_DIR + \"train_images/\"\nTEST_IMG_PATH = KAGGLE_DIR + 'test_images/'","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:50.277651Z","iopub.execute_input":"2022-05-29T20:04:50.278502Z","iopub.status.idle":"2022-05-29T20:04:50.283951Z","shell.execute_reply.started":"2022-05-29T20:04:50.278458Z","shell.execute_reply":"2022-05-29T20:04:50.283202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load csv and convert to pandas dataframe\n\ndf_train = pd.read_csv(TRAIN_DF_PATH)\ndf_train['id_code'] = df_train['id_code'] + \".png\"\n\ndf_test = pd.read_csv(TEST_DF_PATH)\ndf_test['id_code'] = df_test['id_code'] + \".png\"","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:50.285298Z","iopub.execute_input":"2022-05-29T20:04:50.285577Z","iopub.status.idle":"2022-05-29T20:04:50.333075Z","shell.execute_reply.started":"2022-05-29T20:04:50.285548Z","shell.execute_reply":"2022-05-29T20:04:50.332024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Show train dataset shape and informations\n\nprint(\"Train dataset size :\", df_train.shape, \"\\n\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:50.335843Z","iopub.execute_input":"2022-05-29T20:04:50.336203Z","iopub.status.idle":"2022-05-29T20:04:50.347876Z","shell.execute_reply.started":"2022-05-29T20:04:50.336155Z","shell.execute_reply":"2022-05-29T20:04:50.347266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Show test dataset shape and informations\n\nprint(\"Test dataset size :\", df_test.shape, \"\\n\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:50.349029Z","iopub.execute_input":"2022-05-29T20:04:50.34929Z","iopub.status.idle":"2022-05-29T20:04:50.367479Z","shell.execute_reply.started":"2022-05-29T20:04:50.34926Z","shell.execute_reply":"2022-05-29T20:04:50.366253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get number of occurrences of each class in train dataset\n\ndf_train.groupby('diagnosis').count()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:50.369204Z","iopub.execute_input":"2022-05-29T20:04:50.369696Z","iopub.status.idle":"2022-05-29T20:04:50.382842Z","shell.execute_reply.started":"2022-05-29T20:04:50.369637Z","shell.execute_reply":"2022-05-29T20:04:50.382217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom keras import *\nfrom keras.models import Sequential, load_model\n#from keras.optimizers.optimizer_experimental.adam import Adam\nfrom keras.models import Model\nfrom keras.layers import GlobalAveragePooling2D, Dropout\nfrom keras.layers.core import Dense\nimport tensorflow.keras as keras\nfrom tensorflow.keras.applications import EfficientNetB3\n\n# Define some constants for image\n\nIMG_WIDTH = 380\nIMG_HEIGHT = 380\nNUM_DIMENSIONS = 3\nBATCH_SIZE = 1\n\nINPUT_SHAPE = (IMG_WIDTH, IMG_HEIGHT, NUM_DIMENSIONS)\n\n\nefnb3 = EfficientNetB3(weights='imagenet', include_top = False, input_shape = INPUT_SHAPE)\n\n#outputSize = 100\n\nmodel = Sequential()\nmodel.add(efnb3)\nmodel.add(GlobalAveragePooling2D())\n#model.add(Dropout(0.5))\n#model.add(Dense(outputSize))\n\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:04:50.384342Z","iopub.execute_input":"2022-05-29T20:04:50.385081Z","iopub.status.idle":"2022-05-29T20:04:54.958647Z","shell.execute_reply.started":"2022-05-29T20:04:50.385042Z","shell.execute_reply":"2022-05-29T20:04:54.957769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add Image augmentation to our generator, with the parameters we wants to change\n\ntrain_datagen = ImageDataGenerator(rotation_range=15,\n                                   horizontal_flip=False,\n                                   vertical_flip=False,\n                                   rescale=1 / 255.)\n\n# Use the dataframe to define train and validation generators\n\ntrain_generator = train_datagen.flow_from_dataframe(df_train, \n                                                    x_col='id_code', \n                                                    y_col='diagnosis',\n                                                    directory = TRAIN_IMG_PATH,\n                                                    target_size=(IMG_WIDTH, IMG_HEIGHT),\n                                                    batch_size=BATCH_SIZE,\n                                                    class_mode='raw', \n                                                    subset='training')\n","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:49:16.621922Z","iopub.execute_input":"2022-05-29T20:49:16.622679Z","iopub.status.idle":"2022-05-29T20:49:18.162739Z","shell.execute_reply.started":"2022-05-29T20:49:16.622623Z","shell.execute_reply":"2022-05-29T20:49:18.161875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale=1 / 255.)\n\ntrain_generator = train_datagen.flow_from_dataframe(df_test, \n                                                    x_col='id_code', \n                                                    y_col= None,\n                                                    directory = TEST_IMG_PATH,\n                                                    target_size=(IMG_WIDTH, IMG_HEIGHT),\n                                                    batch_size=BATCH_SIZE,\n                                                    class_mode= None)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:11:21.77658Z","iopub.execute_input":"2022-05-30T00:11:21.776951Z","iopub.status.idle":"2022-05-30T00:11:22.529424Z","shell.execute_reply.started":"2022-05-30T00:11:21.776913Z","shell.execute_reply":"2022-05-30T00:11:22.528601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate data with data augmentation\n\ndef generateData(generator, numElements):\n    x = []\n    y = []\n        \n    for j in tqdm(range(0, numElements)):\n        img, label = next(generator)\n        x.append(model.predict(img))\n        y.append(label)\n            \n    return np.array(x),np.array(y)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:11:47.783225Z","iopub.execute_input":"2022-05-30T00:11:47.784124Z","iopub.status.idle":"2022-05-30T00:11:47.789796Z","shell.execute_reply.started":"2022-05-30T00:11:47.784066Z","shell.execute_reply":"2022-05-30T00:11:47.789171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate data with data augmentation\n\ndef generateDataTest(generator, numElements):\n    x = []\n        \n    for j in tqdm(range(0, numElements)):\n        img = next(generator)\n        x.append(model.predict(img))\n            \n    return np.array(x)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:14:16.292543Z","iopub.execute_input":"2022-05-30T00:14:16.292823Z","iopub.status.idle":"2022-05-30T00:14:16.298047Z","shell.execute_reply.started":"2022-05-30T00:14:16.292793Z","shell.execute_reply":"2022-05-30T00:14:16.297193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Generating train dataset!')\nx_train, y_train = generateData(train_generator, 3600)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:36:28.778744Z","iopub.execute_input":"2022-05-29T21:36:28.779786Z","iopub.status.idle":"2022-05-29T21:58:54.006745Z","shell.execute_reply.started":"2022-05-29T21:36:28.779705Z","shell.execute_reply":"2022-05-29T21:58:54.005879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = generateDataTest(train_generator, 1928)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:14:18.19626Z","iopub.execute_input":"2022-05-30T00:14:18.196875Z","iopub.status.idle":"2022-05-30T00:23:15.232428Z","shell.execute_reply.started":"2022-05-30T00:14:18.196826Z","shell.execute_reply":"2022-05-30T00:23:15.231239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = np.array(np.squeeze(x_train, axis = 1))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:58:54.012145Z","iopub.execute_input":"2022-05-29T21:58:54.012383Z","iopub.status.idle":"2022-05-29T21:58:54.021478Z","shell.execute_reply.started":"2022-05-29T21:58:54.012354Z","shell.execute_reply":"2022-05-29T21:58:54.020339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = np.array(np.squeeze(x_test, axis = 1))","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:23:15.235186Z","iopub.execute_input":"2022-05-30T00:23:15.235954Z","iopub.status.idle":"2022-05-30T00:23:15.243018Z","shell.execute_reply.started":"2022-05-30T00:23:15.235892Z","shell.execute_reply":"2022-05-30T00:23:15.242422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\nsm = SMOTE(random_state=42)\nX_res, Y_res = sm.fit_resample(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:58:54.022873Z","iopub.execute_input":"2022-05-29T21:58:54.023236Z","iopub.status.idle":"2022-05-29T21:58:54.412206Z","shell.execute_reply.started":"2022-05-29T21:58:54.023149Z","shell.execute_reply":"2022-05-29T21:58:54.411194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, counts = np.unique(Y_res, return_counts=True)\ndict(zip(unique, counts))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:58:54.4146Z","iopub.execute_input":"2022-05-29T21:58:54.415097Z","iopub.status.idle":"2022-05-29T21:58:54.429363Z","shell.execute_reply.started":"2022-05-29T21:58:54.415065Z","shell.execute_reply":"2022-05-29T21:58:54.428163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_res_2 = np.expand_dims(Y_res, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:58:54.435036Z","iopub.execute_input":"2022-05-29T21:58:54.436595Z","iopub.status.idle":"2022-05-29T21:58:54.440078Z","shell.execute_reply.started":"2022-05-29T21:58:54.436556Z","shell.execute_reply":"2022-05-29T21:58:54.439258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_val, y_train, y_val = train_test_split(X_res, Y_res_2, test_size=0.1, stratify=Y_res_2, shuffle = True, random_state=42)\n#x_train, x_val, y_train, y_val = train_test_split(x_train, y_train, test_size=0.2, shuffle = True, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:58:54.441356Z","iopub.execute_input":"2022-05-29T21:58:54.442039Z","iopub.status.idle":"2022-05-29T21:58:54.511064Z","shell.execute_reply.started":"2022-05-29T21:58:54.441996Z","shell.execute_reply":"2022-05-29T21:58:54.510391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:59:24.137776Z","iopub.execute_input":"2022-05-29T21:59:24.138352Z","iopub.status.idle":"2022-05-29T21:59:24.144995Z","shell.execute_reply.started":"2022-05-29T21:59:24.138291Z","shell.execute_reply":"2022-05-29T21:59:24.144275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-29T21:59:26.557738Z","iopub.execute_input":"2022-05-29T21:59:26.558252Z","iopub.status.idle":"2022-05-29T21:59:26.564835Z","shell.execute_reply.started":"2022-05-29T21:59:26.558191Z","shell.execute_reply":"2022-05-29T21:59:26.563735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:10:36.417177Z","iopub.execute_input":"2022-05-30T00:10:36.417989Z","iopub.status.idle":"2022-05-30T00:10:36.428819Z","shell.execute_reply.started":"2022-05-30T00:10:36.417922Z","shell.execute_reply":"2022-05-30T00:10:36.42768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:42:02.517456Z","iopub.execute_input":"2022-05-29T20:42:02.517662Z","iopub.status.idle":"2022-05-29T20:42:02.530619Z","shell.execute_reply.started":"2022-05-29T20:42:02.517636Z","shell.execute_reply":"2022-05-29T20:42:02.529987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# K-Nearest Neighbors (KNN)","metadata":{}},{"cell_type":"code","source":"'''\nX_train = x_train.reshape((x_train.shape[0], IMG_WIDTH * IMG_HEIGHT* NUM_DIMENSIONS))\nX_val = x_val.reshape((x_val.shape[0], IMG_WIDTH * IMG_HEIGHT* NUM_DIMENSIONS))\n'''","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:42:02.5318Z","iopub.execute_input":"2022-05-29T20:42:02.532662Z","iopub.status.idle":"2022-05-29T20:42:02.545078Z","shell.execute_reply.started":"2022-05-29T20:42:02.532625Z","shell.execute_reply":"2022-05-29T20:42:02.544295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:42:02.54632Z","iopub.execute_input":"2022-05-29T20:42:02.546628Z","iopub.status.idle":"2022-05-29T20:42:02.555397Z","shell.execute_reply.started":"2022-05-29T20:42:02.546596Z","shell.execute_reply":"2022-05-29T20:42:02.554566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train[0].shape","metadata":{"execution":{"iopub.status.busy":"2022-05-29T20:42:02.556594Z","iopub.execute_input":"2022-05-29T20:42:02.556823Z","iopub.status.idle":"2022-05-29T20:42:02.566909Z","shell.execute_reply.started":"2022-05-29T20:42:02.556794Z","shell.execute_reply":"2022-05-29T20:42:02.56626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn_clf = KNeighborsClassifier(n_neighbors=9)\nknn_clf.fit(x_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:15:51.880381Z","iopub.execute_input":"2022-05-29T22:15:51.880995Z","iopub.status.idle":"2022-05-29T22:15:51.899496Z","shell.execute_reply.started":"2022-05-29T22:15:51.880952Z","shell.execute_reply":"2022-05-29T22:15:51.898744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_knn = knn_clf.predict(np.array(x_val))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:15:53.241241Z","iopub.execute_input":"2022-05-29T22:15:53.241853Z","iopub.status.idle":"2022-05-29T22:15:53.889528Z","shell.execute_reply.started":"2022-05-29T22:15:53.241813Z","shell.execute_reply":"2022-05-29T22:15:53.88866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\nfrom sklearn.metrics import accuracy_score\n\nprint(metrics.classification_report(y_val, y_pred_knn))\nprint(accuracy_score(y_val, y_pred_knn))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:15:55.902912Z","iopub.execute_input":"2022-05-29T22:15:55.90325Z","iopub.status.idle":"2022-05-29T22:15:55.918166Z","shell.execute_reply.started":"2022-05-29T22:15:55.903213Z","shell.execute_reply":"2022-05-29T22:15:55.917091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cohen_kappa_score(y_val, y_pred_knn, weights = 'quadratic')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:15:58.972493Z","iopub.execute_input":"2022-05-29T22:15:58.972822Z","iopub.status.idle":"2022-05-29T22:15:58.982503Z","shell.execute_reply.started":"2022-05-29T22:15:58.972782Z","shell.execute_reply":"2022-05-29T22:15:58.981858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decision Tree","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparams = {'max_depth': list(range(10, 20))}\ngrid_search_cv = GridSearchCV(DecisionTreeClassifier(random_state=42), params, verbose=1, cv=3)\n\ngrid_search_cv.fit(x_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:39:54.849902Z","iopub.execute_input":"2022-05-29T22:39:54.850235Z","iopub.status.idle":"2022-05-29T22:45:03.49837Z","shell.execute_reply.started":"2022-05-29T22:39:54.850202Z","shell.execute_reply":"2022-05-29T22:45:03.497364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = grid_search_cv.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:47:00.651002Z","iopub.execute_input":"2022-05-29T22:47:00.651312Z","iopub.status.idle":"2022-05-29T22:47:00.655357Z","shell.execute_reply.started":"2022-05-29T22:47:00.651282Z","shell.execute_reply":"2022-05-29T22:47:00.654408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x['max_depth']","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:47:54.872277Z","iopub.execute_input":"2022-05-29T22:47:54.873053Z","iopub.status.idle":"2022-05-29T22:47:54.879527Z","shell.execute_reply.started":"2022-05-29T22:47:54.873008Z","shell.execute_reply":"2022-05-29T22:47:54.878723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\ntree_clf = DecisionTreeClassifier(max_depth=17, criterion='entropy', random_state=42)\ntree_clf.fit(x_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:45:25.095049Z","iopub.execute_input":"2022-05-29T22:45:25.095384Z","iopub.status.idle":"2022-05-29T22:45:46.251093Z","shell.execute_reply.started":"2022-05-29T22:45:25.09535Z","shell.execute_reply":"2022-05-29T22:45:46.250277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_tree = tree_clf.predict(x_val)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:46:14.195126Z","iopub.execute_input":"2022-05-29T22:46:14.195829Z","iopub.status.idle":"2022-05-29T22:46:14.202196Z","shell.execute_reply.started":"2022-05-29T22:46:14.195787Z","shell.execute_reply":"2022-05-29T22:46:14.201323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\n\nresult = classification_report(y_val, y_pred_tree)\nprint(result)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:46:15.513201Z","iopub.execute_input":"2022-05-29T22:46:15.513525Z","iopub.status.idle":"2022-05-29T22:46:15.530063Z","shell.execute_reply.started":"2022-05-29T22:46:15.513487Z","shell.execute_reply":"2022-05-29T22:46:15.528432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cohen_kappa_score(y_val, y_pred_tree, weights = 'quadratic')","metadata":{"execution":{"iopub.status.busy":"2022-05-29T22:46:22.496609Z","iopub.execute_input":"2022-05-29T22:46:22.496916Z","iopub.status.idle":"2022-05-29T22:46:22.503787Z","shell.execute_reply.started":"2022-05-29T22:46:22.496882Z","shell.execute_reply":"2022-05-29T22:46:22.503238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Support Vector Machines (SVM)","metadata":{}},{"cell_type":"code","source":"params_svm = {'degree': list(range(3, 15))} # Best degree = 14\ngrid_search_cv = GridSearchCV(svm.SVC(kernel = 'poly', random_state=42), params_svm, verbose=1, cv=3)\n\ngrid_search_cv.fit(x_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T23:39:40.92892Z","iopub.execute_input":"2022-05-29T23:39:40.929408Z","iopub.status.idle":"2022-05-30T00:06:31.194537Z","shell.execute_reply.started":"2022-05-29T23:39:40.929372Z","shell.execute_reply":"2022-05-30T00:06:31.193516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search_cv.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:08:00.68042Z","iopub.execute_input":"2022-05-30T00:08:00.680775Z","iopub.status.idle":"2022-05-30T00:08:00.688462Z","shell.execute_reply.started":"2022-05-30T00:08:00.680737Z","shell.execute_reply":"2022-05-30T00:08:00.687849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import svm\n\nsvm_clf = svm.SVC(kernel = 'poly', degree = 14)\nsvm_clf.fit(x_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:08:04.451573Z","iopub.execute_input":"2022-05-30T00:08:04.452441Z","iopub.status.idle":"2022-05-30T00:09:08.221269Z","shell.execute_reply.started":"2022-05-30T00:08:04.452404Z","shell.execute_reply":"2022-05-30T00:09:08.220317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_svm = svm_clf.predict(x_val)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:09:08.223327Z","iopub.execute_input":"2022-05-30T00:09:08.223799Z","iopub.status.idle":"2022-05-30T00:09:14.318809Z","shell.execute_reply.started":"2022-05-30T00:09:08.223751Z","shell.execute_reply":"2022-05-30T00:09:14.318049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_svm = metrics.classification_report(y_val, y_pred_svm)\nprint(result_svm)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:09:14.320062Z","iopub.execute_input":"2022-05-30T00:09:14.320298Z","iopub.status.idle":"2022-05-30T00:09:14.333985Z","shell.execute_reply.started":"2022-05-30T00:09:14.320272Z","shell.execute_reply":"2022-05-30T00:09:14.333201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##Naive Bayes","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\nsvm_nb = GaussianNB(var_smoothing = 1e-4)\nsvm_nb.fit(x_train, np.ravel(y_train))\ny_pred_nb = svm_nb.predict(x_val)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T23:31:20.552195Z","iopub.execute_input":"2022-05-29T23:31:20.552506Z","iopub.status.idle":"2022-05-29T23:31:20.661051Z","shell.execute_reply.started":"2022-05-29T23:31:20.552468Z","shell.execute_reply":"2022-05-29T23:31:20.660329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_nb = metrics.classification_report(y_val, y_pred_nb)\nprint(result_nb)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T23:31:21.894813Z","iopub.execute_input":"2022-05-29T23:31:21.895371Z","iopub.status.idle":"2022-05-29T23:31:21.906792Z","shell.execute_reply.started":"2022-05-29T23:31:21.895336Z","shell.execute_reply":"2022-05-29T23:31:21.906178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LogisticRegression","metadata":{}},{"cell_type":"code","source":"params_lg = {'tol': [1e-8, 1e-6, 1e-5], 'C': [0.1, 1 ,10 ,100]}\ngrid_search_cv = GridSearchCV(LogisticRegression(random_state=42, max_iter= 500), params_lg, verbose=1, cv=3)\n\ngrid_search_cv.fit(x_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-05-29T23:16:58.229591Z","iopub.execute_input":"2022-05-29T23:16:58.230601Z","iopub.status.idle":"2022-05-29T23:18:32.003533Z","shell.execute_reply.started":"2022-05-29T23:16:58.230544Z","shell.execute_reply":"2022-05-29T23:18:32.002601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_search_cv.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-05-29T23:22:02.421792Z","iopub.execute_input":"2022-05-29T23:22:02.422167Z","iopub.status.idle":"2022-05-29T23:22:02.429468Z","shell.execute_reply.started":"2022-05-29T23:22:02.42213Z","shell.execute_reply":"2022-05-29T23:22:02.428765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlg_clf = LogisticRegression(random_state=42, C = 10000, max_iter = 10000).fit(x_train, np.ravel(y_train))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_lg = lg_clf.predict(x_val)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:14:05.284173Z","iopub.status.idle":"2022-05-30T00:14:05.285Z","shell.execute_reply.started":"2022-05-30T00:14:05.284699Z","shell.execute_reply":"2022-05-30T00:14:05.284731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_lg = metrics.classification_report(y_val, y_pred_lg)\nprint(result_lg)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:14:05.286565Z","iopub.status.idle":"2022-05-30T00:14:05.287864Z","shell.execute_reply.started":"2022-05-30T00:14:05.287575Z","shell.execute_reply":"2022-05-30T00:14:05.287615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"y_pred_tree_sub = tree_clf.predict(x_test)\n\nlabels_tree = np.array(y_pred_tree_sub)\nsubmission_df = pd.DataFrame({'id_code':df_test['id_code'],'diagnosis':labels_tree}) #DataFrame with id_code's and predicted labels.\n\nsubmission_df.to_csv('submission.csv',index=False) #csv file for submission.","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:25:46.498904Z","iopub.execute_input":"2022-05-30T00:25:46.499442Z","iopub.status.idle":"2022-05-30T00:25:46.530009Z","shell.execute_reply.started":"2022-05-30T00:25:46.4994Z","shell.execute_reply":"2022-05-30T00:25:46.529174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2022-05-30T00:25:47.964021Z","iopub.execute_input":"2022-05-30T00:25:47.964357Z","iopub.status.idle":"2022-05-30T00:25:47.979128Z","shell.execute_reply.started":"2022-05-30T00:25:47.964322Z","shell.execute_reply":"2022-05-30T00:25:47.978152Z"},"trusted":true},"execution_count":null,"outputs":[]}]}