{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#Importing packages\nimport tensorflow as tf\nimport os\nimport cv2\nimport imageio\nimport numpy as np\nfrom tqdm import tqdm\nimport seaborn as sns\nimport matplotlib.patches as mpatches\nfrom scipy.signal import find_peaks\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import Dense,Input,Conv2D,MaxPool2D,Activation,Dropout,Flatten\nfrom tensorflow.keras.models import Model\nimport random as rn\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.applications import VGG16\nimport datetime\nimport glob\nimport warnings\nfrom tensorflow.keras import models, layers\nfrom sklearn.metrics import cohen_kappa_score\nimport math\nfrom keras.regularizers import l1 ,l2\nimport keras\nfrom tensorflow.keras.layers import BatchNormalization, Activation, Flatten\nfrom tensorflow.keras.optimizers import Adam, SGD\nfrom sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n\nimport argparse\nimport os\nimport warnings\n\nfrom keras.callbacks import Callback\nfrom keras import backend as K\nwarnings.filterwarnings('ignore')\n\n\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-03T04:11:26.705191Z","iopub.execute_input":"2021-09-03T04:11:26.705582Z","iopub.status.idle":"2021-09-03T04:11:31.896385Z","shell.execute_reply.started":"2021-09-03T04:11:26.705496Z","shell.execute_reply":"2021-09-03T04:11:31.895556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nimg_path = list()\nimg_fullname = list()\nimg_name = list()\nfor dirname, _, filenames in os.walk('/kaggle/input/aptos2019-blindness-detection/train_images/'):\n    for filename in filenames:\n        img_path.append(os.path.join(dirname, filename))\n        temp = os.path.join(dirname, filename)\n        temp = temp.split(\"/\")[-1]\n        img_fullname.append(temp)\n        temp = temp.split(\".\")[0]\n        img_name.append(str(temp))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:16.128325Z","iopub.execute_input":"2021-09-03T04:13:16.128674Z","iopub.status.idle":"2021-09-03T04:13:19.681781Z","shell.execute_reply.started":"2021-09-03T04:13:16.12864Z","shell.execute_reply":"2021-09-03T04:13:19.680934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1.1 Reading data**","metadata":{}},{"cell_type":"code","source":"pre_df = pd.DataFrame()\npre_df['path'] = img_path\npre_df['fullname'] = img_fullname\npre_df['id_code'] = img_name","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:21.598844Z","iopub.execute_input":"2021-09-03T04:13:21.599148Z","iopub.status.idle":"2021-09-03T04:13:21.615015Z","shell.execute_reply.started":"2021-09-03T04:13:21.599119Z","shell.execute_reply":"2021-09-03T04:13:21.61399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading DR grades files of 2019, Messidor & IDRiD\ntrain2019 = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntrain2019 = train2019[['id_code', 'diagnosis']]","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:27.925352Z","iopub.execute_input":"2021-09-03T04:13:27.925715Z","iopub.status.idle":"2021-09-03T04:13:27.945761Z","shell.execute_reply.started":"2021-09-03T04:13:27.925679Z","shell.execute_reply":"2021-09-03T04:13:27.94499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetching DR grades of the respective images of 2019\nlabels = list()\nid_code = list()\nfor i, j in train2019.iterrows():\n    id_code.append(j['id_code'])\n    labels.append(j['diagnosis'])","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:33.425979Z","iopub.execute_input":"2021-09-03T04:13:33.426298Z","iopub.status.idle":"2021-09-03T04:13:33.699914Z","shell.execute_reply.started":"2021-09-03T04:13:33.426267Z","shell.execute_reply":"2021-09-03T04:13:33.699102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.DataFrame() # Storing the image file name & DR grades into the dataframe\ntrain_labels['id_code'] = id_code\ntrain_labels['diagnosis'] = labels","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:37.763724Z","iopub.execute_input":"2021-09-03T04:13:37.764069Z","iopub.status.idle":"2021-09-03T04:13:37.773217Z","shell.execute_reply.started":"2021-09-03T04:13:37.764036Z","shell.execute_reply":"2021-09-03T04:13:37.772301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merging the image path & DR grades \npre_df = pd.merge(pre_df, train_labels, on='id_code', how='left')","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:42.891055Z","iopub.execute_input":"2021-09-03T04:13:42.891369Z","iopub.status.idle":"2021-09-03T04:13:42.909801Z","shell.execute_reply.started":"2021-09-03T04:13:42.891339Z","shell.execute_reply":"2021-09-03T04:13:42.909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pre_df = pre_df[pre_df['diagnosis'].notna()] # removing row if the DR grade is NaN","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:13:47.865704Z","iopub.execute_input":"2021-09-03T04:13:47.866017Z","iopub.status.idle":"2021-09-03T04:13:47.874621Z","shell.execute_reply.started":"2021-09-03T04:13:47.865987Z","shell.execute_reply":"2021-09-03T04:13:47.873705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(pre_df[pre_df['diagnosis'].isnull()]) # Verify if there is any NaN present in the DR grades field","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:14:00.568972Z","iopub.execute_input":"2021-09-03T04:14:00.569287Z","iopub.status.idle":"2021-09-03T04:14:00.579874Z","shell.execute_reply.started":"2021-09-03T04:14:00.569255Z","shell.execute_reply":"2021-09-03T04:14:00.578709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert_dict = {'diagnosis': str}\n  \npre_df = pre_df.astype(convert_dict)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:14:06.539634Z","iopub.execute_input":"2021-09-03T04:14:06.539951Z","iopub.status.idle":"2021-09-03T04:14:06.551586Z","shell.execute_reply.started":"2021-09-03T04:14:06.539924Z","shell.execute_reply":"2021-09-03T04:14:06.549304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train and test data split up**","metadata":{}},{"cell_type":"code","source":"# train test split\nfrom sklearn.model_selection import train_test_split\ny = pre_df['diagnosis'].values\nX_train, X_test, y_train, y_test = train_test_split(pre_df, y, test_size=0.20, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:14:45.789661Z","iopub.execute_input":"2021-09-03T04:14:45.789984Z","iopub.status.idle":"2021-09-03T04:14:45.805898Z","shell.execute_reply.started":"2021-09-03T04:14:45.789954Z","shell.execute_reply":"2021-09-03T04:14:45.804895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)\nprint(y_train.shape)\nprint(y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:14:51.705318Z","iopub.execute_input":"2021-09-03T04:14:51.705697Z","iopub.status.idle":"2021-09-03T04:14:51.710767Z","shell.execute_reply.started":"2021-09-03T04:14:51.705665Z","shell.execute_reply":"2021-09-03T04:14:51.709922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reference URL - https://vijayabhaskar96.medium.com/tutorial-on-keras-flow-from-dataframe-1fd4493d237c\ntf.keras.backend.clear_session()\nfrom keras_preprocessing.image import ImageDataGenerator\n\ndatagen=ImageDataGenerator(rescale=1./255., validation_split=0.20, rotation_range=15, fill_mode='nearest', width_shift_range=0.1, height_shift_range=0.1, \n                horizontal_flip=True, vertical_flip=True)\n\ntrain_generator=datagen.flow_from_dataframe(dataframe=X_train, directory=None, x_col=\"path\", y_col=\"diagnosis\", subset=\"training\", \n                batch_size=8, seed=10, shuffle=True, class_mode=\"categorical\", drop_duplicates = False, target_size=(320,320))\n\nvalid_generator=datagen.flow_from_dataframe(dataframe=X_train, directory=None, x_col=\"path\", y_col=\"diagnosis\", subset=\"validation\", \n                        batch_size=8, seed=10, shuffle=True, class_mode=\"categorical\", drop_duplicates = False, target_size=(320,320))\n\ntest_datagen=ImageDataGenerator(rescale=1./255.)\n\ntest_generator=test_datagen.flow_from_dataframe(dataframe=X_test, directory=None, x_col=\"path\", \n                y_col=None, batch_size=1,shuffle=False, class_mode=None, target_size=(320,320))","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:15:05.455826Z","iopub.execute_input":"2021-09-03T04:15:05.45614Z","iopub.status.idle":"2021-09-03T04:15:07.117999Z","shell.execute_reply.started":"2021-09-03T04:15:05.456109Z","shell.execute_reply":"2021-09-03T04:15:07.117058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building VGG16 Model**","metadata":{}},{"cell_type":"code","source":"os.environ['PYTHONHASHSEED'] = '0'\n\n##https://keras.io/getting-started/faq/#how-can-i-obtain-reproducible-results-using-keras-during-development\n## Have to clear the session. If you are not clearing, Graph will create again and again and graph size will increses. \n## Varibles will also set to some value from before session\ntf.keras.backend.clear_session()\n\n## Set the random seed values to regenerate the model.\ntf.keras.backend.clear_session()\nnp.random.seed(0)\nrn.seed(0)\n\nvgg_base = VGG16(include_top=False, weights=None, input_shape=(320,320,3))\n\n# Freeze base model\nvgg_base.trainable = False\n\n\n#Input layer\ninput_layer = Input(shape=(320,320,3))\n\nvgg_layer= vgg_base(input_layer)\n\n#Conv Layer\nConv1 = Conv2D(128, 10, strides=1 , padding = 'valid', activation='relu')(vgg_layer)\n\n\n#MaxPool Layer\nConv2 = Conv2D(64, 1, strides=1 , padding = 'valid', activation='relu')(Conv1)\n\nflatten = Flatten(data_format='channels_last')(Conv2)\n#output layer\noutputs = Dense(5,activation='softmax')(flatten)\n\n\n#Creating a model\nvgg_model = Model(inputs=input_layer,outputs=outputs)\nvgg_model.load_weights('../input/vgg-pretrained-weights/weights-vgg.hdf5')\n\n#creating object to ReduceLROnPlateau class to decay lr by 10% If validation accuracy is less than previous epoch accuracy(cond1)\nreducelrate = tf.keras.callbacks.ReduceLROnPlateau(monitor='accuracy', factor=0.96, verbose=1, patience=2, cooldown=2)\n\ncallbacks_list = [reducelrate] #Merging all the callbacks in a list\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.0005)\n\nvgg_model.compile(optimizer=optimizer, loss='categorical_crossentropy',metrics=['accuracy']) #Compiling model with cross-entropy as loss\nprint(vgg_model.summary())\n","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:15:29.977312Z","iopub.execute_input":"2021-09-03T04:15:29.977823Z","iopub.status.idle":"2021-09-03T04:15:35.184398Z","shell.execute_reply.started":"2021-09-03T04:15:29.977782Z","shell.execute_reply":"2021-09-03T04:15:35.183596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training the VGG16 model**","metadata":{}},{"cell_type":"code","source":"STEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\nSTEP_SIZE_TEST=test_generator.n//test_generator.batch_size\nvgg_model.fit_generator(generator=train_generator, steps_per_epoch=STEP_SIZE_TRAIN, validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID, epochs=80, callbacks=callbacks_list)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T04:16:13.198994Z","iopub.execute_input":"2021-09-03T04:16:13.199321Z","iopub.status.idle":"2021-09-03T06:17:17.761973Z","shell.execute_reply.started":"2021-09-03T04:16:13.19929Z","shell.execute_reply":"2021-09-03T06:17:17.761105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Prediction on Test images**","metadata":{}},{"cell_type":"code","source":"y_pred = vgg_model.predict(test_generator)\ny_pred_class = np.argmax(y_pred, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:30:46.774027Z","iopub.execute_input":"2021-09-03T10:30:46.774342Z","iopub.status.idle":"2021-09-03T10:32:04.494243Z","shell.execute_reply.started":"2021-09-03T10:30:46.774314Z","shell.execute_reply":"2021-09-03T10:32:04.493419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Calculating Quadratic cohen kappa score**","metadata":{}},{"cell_type":"code","source":"cohen = cohen_kappa_score(y_pred_class, y_test.astype('int'), weights='quadratic')\nprint(\"Quadratic Cohen kappa score of VGG model on test data is - %.3f\" %cohen)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:32:37.863596Z","iopub.execute_input":"2021-09-03T10:32:37.863917Z","iopub.status.idle":"2021-09-03T10:32:37.872223Z","shell.execute_reply.started":"2021-09-03T10:32:37.863888Z","shell.execute_reply":"2021-09-03T10:32:37.871226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Calculating Accuracy & Macro F1 score on test data**","metadata":{}},{"cell_type":"code","source":"print ('Classification Report : \\n', classification_report(y_test.astype('int'), y_pred_class))\nsns.heatmap(confusion_matrix(y_test.astype('int'), y_pred_class), annot=True, fmt=\"d\");\nplt.title(\"Confusion matrix\")\nplt.ylabel('Actual class')\nplt.xlabel('Predicted class')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:32:46.411294Z","iopub.execute_input":"2021-09-03T10:32:46.41166Z","iopub.status.idle":"2021-09-03T10:32:46.724924Z","shell.execute_reply.started":"2021-09-03T10:32:46.411628Z","shell.execute_reply":"2021-09-03T10:32:46.724014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Reading Test data**","metadata":{}},{"cell_type":"code","source":"img_path = list()\nimg_fullname = list()\nimg_name = list()\nfor dirname, _, filenames in os.walk('../input/aptos2019-blindness-detection/test_images/'):\n    for filename in filenames:\n        img_path.append(os.path.join(dirname, filename))\n        temp = os.path.join(dirname, filename)\n        temp = temp.split(\"/\")[-1]\n        img_fullname.append(temp)\n        temp = temp.split(\".\")[0]\n        img_name.append(str(temp))\n","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:33:23.803246Z","iopub.execute_input":"2021-09-03T10:33:23.803574Z","iopub.status.idle":"2021-09-03T10:33:25.515058Z","shell.execute_reply.started":"2021-09-03T10:33:23.803545Z","shell.execute_reply":"2021-09-03T10:33:25.514204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pre_df = pd.DataFrame()\npre_df['path'] = img_path\npre_df['fullname'] = img_fullname\npre_df['id_code'] = img_name","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:36:20.780786Z","iopub.execute_input":"2021-09-03T10:36:20.781119Z","iopub.status.idle":"2021-09-03T10:36:20.792237Z","shell.execute_reply.started":"2021-09-03T10:36:20.781089Z","shell.execute_reply":"2021-09-03T10:36:20.791457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen=ImageDataGenerator(rescale=1./255.)\n\ntest_generator_2019=test_datagen.flow_from_dataframe(dataframe=pre_df, directory=None, x_col=\"path\", \n                y_col=None, batch_size=1, shuffle=False, class_mode=None, target_size=(320,320))","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:37:41.882301Z","iopub.execute_input":"2021-09-03T10:37:41.882654Z","iopub.status.idle":"2021-09-03T10:37:42.56526Z","shell.execute_reply.started":"2021-09-03T10:37:41.882624Z","shell.execute_reply":"2021-09-03T10:37:42.564351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_2019 = vgg_model.predict(test_generator_2019)\ny_pred_class_2019 = np.argmax(y_pred_2019, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:37:54.785061Z","iopub.execute_input":"2021-09-03T10:37:54.785387Z","iopub.status.idle":"2021-09-03T10:39:45.00499Z","shell.execute_reply.started":"2021-09-03T10:37:54.785357Z","shell.execute_reply":"2021-09-03T10:39:45.004153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = list()\nfor i in test_generator_2019.filenames:\n    temp = i.split('.')[-2]\n    temp = temp.split('/')[-1]\n    filenames.append(temp)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:41:37.320776Z","iopub.execute_input":"2021-09-03T10:41:37.321092Z","iopub.status.idle":"2021-09-03T10:41:37.329137Z","shell.execute_reply.started":"2021-09-03T10:41:37.321061Z","shell.execute_reply":"2021-09-03T10:41:37.32815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_vgg = pd.DataFrame()\nsubmission_vgg['id_code'] = filenames\nsubmission_vgg['diagnosis'] = y_pred_class_2019","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:41:43.55695Z","iopub.execute_input":"2021-09-03T10:41:43.557252Z","iopub.status.idle":"2021-09-03T10:41:43.565568Z","shell.execute_reply.started":"2021-09-03T10:41:43.557224Z","shell.execute_reply":"2021-09-03T10:41:43.564434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_vgg","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:41:53.086929Z","iopub.execute_input":"2021-09-03T10:41:53.087382Z","iopub.status.idle":"2021-09-03T10:41:53.09864Z","shell.execute_reply.started":"2021-09-03T10:41:53.087346Z","shell.execute_reply":"2021-09-03T10:41:53.097512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_vgg.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-03T10:42:12.983865Z","iopub.execute_input":"2021-09-03T10:42:12.984171Z","iopub.status.idle":"2021-09-03T10:42:12.995796Z","shell.execute_reply.started":"2021-09-03T10:42:12.984142Z","shell.execute_reply":"2021-09-03T10:42:12.994999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}