{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2022-09-30T09:47:32.209784Z","iopub.execute_input":"2022-09-30T09:47:32.210221Z","iopub.status.idle":"2022-09-30T09:47:40.925128Z","shell.execute_reply.started":"2022-09-30T09:47:32.210131Z","shell.execute_reply":"2022-09-30T09:47:40.924153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.layers import Input, Lambda, Dense, Flatten, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.optimizers import Adam, RMSprop, SGD\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.inception_resnet_v2 import InceptionResNetV2\nfrom tensorflow.keras.applications.inception_resnet_v2 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img\nfrom tensorflow.keras.models import Sequential\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:47:43.864322Z","iopub.execute_input":"2022-09-30T09:47:43.864676Z","iopub.status.idle":"2022-09-30T09:47:49.316526Z","shell.execute_reply.started":"2022-09-30T09:47:43.864647Z","shell.execute_reply":"2022-09-30T09:47:49.315482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom keras.callbacks import Callback, ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom sklearn.metrics import cohen_kappa_score\n\ndef kappa(y_true, y_pred):\n  y_true = np.argmax(y_true, axis=-1)\n  y_pred = np.argmax(y_pred, axis=-1)\n  kappa_score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n  return kappa_score\n\ndef kappa_metric(y_true, y_pred):\n  kappa_score = tf.py_function(func=kappa, inp=[y_true, y_pred], Tout=tf.float32)\n  return kappa_score","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:47:49.318346Z","iopub.execute_input":"2022-09-30T09:47:49.318934Z","iopub.status.idle":"2022-09-30T09:47:49.838903Z","shell.execute_reply.started":"2022-09-30T09:47:49.318892Z","shell.execute_reply":"2022-09-30T09:47:49.837708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Inception V3**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionV3","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:48:30.355802Z","iopub.execute_input":"2022-09-30T09:48:30.356483Z","iopub.status.idle":"2022-09-30T09:48:30.361093Z","shell.execute_reply.started":"2022-09-30T09:48:30.356443Z","shell.execute_reply":"2022-09-30T09:48:30.360077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = [299, 299]\n\ninceptionv3 = InceptionV3(input_shape=IMAGE_SIZE + [3], weights='imagenet', include_top=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:48:33.994761Z","iopub.execute_input":"2022-09-30T09:48:33.995145Z","iopub.status.idle":"2022-09-30T09:48:39.975298Z","shell.execute_reply.started":"2022-09-30T09:48:33.995114Z","shell.execute_reply":"2022-09-30T09:48:39.974181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in inceptionv3.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-09-30T05:56:16.376949Z","iopub.execute_input":"2022-09-30T05:56:16.377327Z","iopub.status.idle":"2022-09-30T05:56:16.391786Z","shell.execute_reply.started":"2022-09-30T05:56:16.377294Z","shell.execute_reply":"2022-09-30T05:56:16.389756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = GlobalAveragePooling2D()(inceptionv3.output)\nx = Dropout(0.5)(x)\nx = Dense(2048, activation='relu')(x)\nx = Dropout(0.5)(x)\nx = Flatten()(x)\noutput = Dense(5, activation='softmax', name='final_output')(x)\n\n# create a model object\niv3 = Model(inputs=inceptionv3.input, outputs=output)\niv3.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-09-30T09:48:41.302361Z","iopub.execute_input":"2022-09-30T09:48:41.303398Z","iopub.status.idle":"2022-09-30T09:48:41.384070Z","shell.execute_reply.started":"2022-09-30T09:48:41.303357Z","shell.execute_reply":"2022-09-30T09:48:41.383004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tell the model what cost and optimization method to use\niv3.compile(\n  loss='categorical_crossentropy',\n  optimizer= Adam(0.0001),\n  metrics=['accuracy', kappa_metric]\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:49:03.756331Z","iopub.execute_input":"2022-09-30T09:49:03.756695Z","iopub.status.idle":"2022-09-30T09:49:03.776326Z","shell.execute_reply.started":"2022-09-30T09:49:03.756664Z","shell.execute_reply":"2022-09-30T09:49:03.775245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Creating Image Paths in csv**","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntrain_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['diagnosis'] = train_df['diagnosis'].astype('string')\ntrain_df['fullpath'] = '../input/aptos-processed-training-images/preprocess/' + train_df['id_code'] + '.png'\ntrain_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\ntest_df['fullpath'] = '../input/aptos2019-blindness-detection/test_images/' + test_df['id_code'] + '.png'\ntest_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Image Data Generation**","metadata":{}},{"cell_type":"code","source":"# Use the Image Data Generator to import the images from the dataset\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntrain_df2, valid_df = train_test_split(train_df, train_size=0.8, shuffle=True, random_state=42)\n\ntrain_datagen = ImageDataGenerator(rescale = 1./255,\n                                   shear_range = 0.2,\n                                   zoom_range = 0.2,\n                                   horizontal_flip = True)\n\nvalid_datagen = ImageDataGenerator(rescale = 1./255,\n                                   shear_range = 0.2,\n                                   zoom_range = 0.2,\n                                   horizontal_flip = True)\n\ntest_datagen = ImageDataGenerator(rescale = 1./255)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:49:12.586757Z","iopub.execute_input":"2022-09-30T09:49:12.587154Z","iopub.status.idle":"2022-09-30T09:49:12.598580Z","shell.execute_reply.started":"2022-09-30T09:49:12.587122Z","shell.execute_reply":"2022-09-30T09:49:12.597616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_col = 'fullpath'\ny_col = 'diagnosis'\nimage_shape = (299, 299)\nclass_mode = 'categorical'\nbatch_size = 32\n\ntrain_set = train_datagen.flow_from_dataframe(train_df2, x_col=x_col, y_col=y_col, target_size=image_shape,\n                                              class_mode=class_mode, batch_size=batch_size, shuffle=True,\n                                              random_state=42)\n\nvalid_set = valid_datagen.flow_from_dataframe(valid_df, x_col=x_col, y_col=y_col, target_size=image_shape,\n                                              class_mode=class_mode, batch_size=batch_size, shuffle=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:49:14.417699Z","iopub.execute_input":"2022-09-30T09:49:14.418088Z","iopub.status.idle":"2022-09-30T09:49:16.143546Z","shell.execute_reply.started":"2022-09-30T09:49:14.418052Z","shell.execute_reply":"2022-09-30T09:49:16.142576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = './inceptionv3.h5'\ncheckpoint = ModelCheckpoint(filepath, monitor='val_kappa_metric', verbose=1, save_best_only=True, mode='max')\ncallbacks_list = [checkpoint]","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:49:30.275275Z","iopub.execute_input":"2022-09-30T09:49:30.275646Z","iopub.status.idle":"2022-09-30T09:49:30.281575Z","shell.execute_reply.started":"2022-09-30T09:49:30.275615Z","shell.execute_reply":"2022-09-30T09:49:30.280352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Training the Model**","metadata":{}},{"cell_type":"code","source":"# fit the model\n# Run the cell. It will take some time to execute\niv3_history = iv3.fit(\n  train_set,\n  validation_data=valid_set,\n  epochs=50,\n  steps_per_epoch=len(train_set),\n  validation_steps=len(valid_set),\n  callbacks=callbacks_list\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T09:50:09.631337Z","iopub.execute_input":"2022-09-30T09:50:09.631699Z","iopub.status.idle":"2022-09-30T11:19:09.720810Z","shell.execute_reply.started":"2022-09-30T09:50:09.631668Z","shell.execute_reply":"2022-09-30T11:19:09.719348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a href='./inceptionv3.h5'>Download Inception V3 Model </a>","metadata":{}},{"cell_type":"code","source":"# plot the loss\nplt.plot(iv3_history.history['loss'], label='train loss')\nplt.plot(iv3_history.history['val_loss'], label='val loss')\nplt.legend()\nplt.show()\n\n# plot the accuracy\nplt.plot(iv3_history.history['accuracy'], label='train acc')\nplt.plot(iv3_history.history['val_accuracy'], label='val acc')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T11:20:02.182859Z","iopub.execute_input":"2022-09-30T11:20:02.183999Z","iopub.status.idle":"2022-09-30T11:20:02.625912Z","shell.execute_reply.started":"2022-09-30T11:20:02.183946Z","shell.execute_reply":"2022-09-30T11:20:02.624910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"complete_datagen = ImageDataGenerator(rescale=1./255)\ncomplete_generator = complete_datagen.flow_from_dataframe(dataframe=train_df,\n                                                          x_col=\"fullpath\",\n                                                          target_size=(299,299),\n                                                          batch_size=1,\n                                                          shuffle=False,\n                                                          class_mode=None)\n\nSTEP_SIZE_COMPLETE = complete_generator.n//complete_generator.batch_size\ntrain_preds = iv3.predict_generator(complete_generator, steps=STEP_SIZE_COMPLETE,verbose = 1)\ntrain_preds = [np.argmax(pred) for pred in train_preds]\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T11:20:30.332811Z","iopub.execute_input":"2022-09-30T11:20:30.333867Z","iopub.status.idle":"2022-09-30T11:21:43.552422Z","shell.execute_reply.started":"2022-09-30T11:20:30.333816Z","shell.execute_reply":"2022-09-30T11:21:43.551497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, cohen_kappa_score,accuracy_score, f1_score, classification_report\nprint(\"Train Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_df['diagnosis'].astype('int'), weights='quadratic'))\nprint(\"Train Accuracy score : %.3f\" % accuracy_score(train_df['diagnosis'].astype('int'),train_preds))\nprint(\"Train F1-Score : %.3f\" % f1_score(train_df['diagnosis'].astype('int'),train_preds, average='weighted'))\nprint('\\n')\nprint(classification_report(train_df['diagnosis'].astype('int'),train_preds))","metadata":{"execution":{"iopub.status.busy":"2022-09-30T11:31:02.403558Z","iopub.execute_input":"2022-09-30T11:31:02.403961Z","iopub.status.idle":"2022-09-30T11:31:02.429413Z","shell.execute_reply.started":"2022-09-30T11:31:02.403908Z","shell.execute_reply":"2022-09-30T11:31:02.428363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nlabels = ['0', '1', '2', '3', '4']\ncnf_matrix = confusion_matrix(train_df['diagnosis'].astype('int'), train_preds)\ncnf_matrix_norm = cnf_matrix.astype('float') / cnf_matrix.sum(axis=1)[:, np.newaxis]\ndf_cm = pd.DataFrame(cnf_matrix_norm, index=labels, columns=labels)\nplt.figure(figsize=(16, 7))\n\nsns.heatmap(df_cm, annot=True, fmt='.2f', cmap=\"Blues\")\nplt.xlabel('Predicted Class')\nplt.ylabel('Original Class')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T11:32:12.849619Z","iopub.execute_input":"2022-09-30T11:32:12.850000Z","iopub.status.idle":"2022-09-30T11:32:13.140952Z","shell.execute_reply.started":"2022-09-30T11:32:12.849964Z","shell.execute_reply":"2022-09-30T11:32:13.139902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Validation Scores","metadata":{}},{"cell_type":"code","source":"valid_set = valid_datagen.flow_from_dataframe(dataframe=valid_df,\n                                                          x_col=\"fullpath\",\n                                                          target_size=(299,299),\n                                                          batch_size=1,\n                                                          shuffle=False,\n                                                          class_mode=None)\n\n\nSTEP_SIZE_COMPLETE = valid_set.n//valid_set.batch_size\nvalid_preds = iv3.predict_generator(valid_set, steps=STEP_SIZE_COMPLETE,verbose = 1)\nvalid_preds = [np.argmax(pred) for pred in valid_preds]","metadata":{"execution":{"iopub.status.busy":"2022-09-30T11:49:01.973627Z","iopub.execute_input":"2022-09-30T11:49:01.974087Z","iopub.status.idle":"2022-09-30T11:49:43.505488Z","shell.execute_reply.started":"2022-09-30T11:49:01.974050Z","shell.execute_reply":"2022-09-30T11:49:43.504441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, cohen_kappa_score,accuracy_score, f1_score, classification_report\nprint(\"Train Cohen Kappa score: %.3f\" % cohen_kappa_score(valid_preds, valid_df['diagnosis'].astype('int'), weights='quadratic'))\nprint(\"Train Accuracy score : %.3f\" % accuracy_score(valid_df['diagnosis'].astype('int'),valid_preds))\nprint(\"Train F1-Score : %.3f\" % f1_score(valid_df['diagnosis'].astype('int'),valid_preds, average='weighted'))\nprint('\\n')\nprint(classification_report(valid_df['diagnosis'].astype('int'), valid_preds))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-30T11:51:40.801529Z","iopub.execute_input":"2022-09-30T11:51:40.802630Z","iopub.status.idle":"2022-09-30T11:51:40.823527Z","shell.execute_reply.started":"2022-09-30T11:51:40.802583Z","shell.execute_reply":"2022-09-30T11:51:40.822595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, layer in enumerate(iv3.layers):\n    print(i, layer)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T07:08:54.278219Z","iopub.execute_input":"2022-09-30T07:08:54.278728Z","iopub.status.idle":"2022-09-30T07:08:54.295343Z","shell.execute_reply.started":"2022-09-30T07:08:54.278685Z","shell.execute_reply":"2022-09-30T07:08:54.293942Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fine-Tuning\n\niv3.load_weights('../input/models/inceptionv3.h5')\n\nfor layer in iv3.layers[:249]:\n    layer.trainable = False\n\nfor layer in iv3.layers[249:]:\n    layer.trainable = True\n    \niv3.compile(optimizer=RMSprop(0.00001),\n           loss='categorical_crossentropy',\n           metrics=['accuracy', kappa_metric])\n\ncheckpoint = ModelCheckpoint('./halftuned_final_inceptionv3.h5', monitor='val_kappa_metric', verbose=1, save_best_only=True, mode='max')\nes = EarlyStopping(monitor='val_loss', mode='min', patience=5, restore_best_weights=True, verbose=1)\nrlrop = ReduceLROnPlateau(monitor='val_loss', mode='min', patience=3, factor=0.5, min_lr=1e-7, verbose=1)\n\ncallbacks_list = [checkpoint, es, rlrop]\n\niv3_history = iv3.fit(\n  train_set,\n  validation_data=valid_set,\n  epochs=50,\n  steps_per_epoch=len(train_set),\n  validation_steps=len(valid_set),\n  callbacks=callbacks_list\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-30T07:11:55.404304Z","iopub.execute_input":"2022-09-30T07:11:55.405029Z","iopub.status.idle":"2022-09-30T07:47:58.573070Z","shell.execute_reply.started":"2022-09-30T07:11:55.404992Z","shell.execute_reply":"2022-09-30T07:47:58.571937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the loss\nplt.plot(iv3_history.history['loss'], label='train loss')\nplt.plot(iv3_history.history['val_loss'], label='val loss')\nplt.legend()\nplt.show()\n\n# plot the accuracy\nplt.plot(iv3_history.history['accuracy'], label='train acc')\nplt.plot(iv3_history.history['val_accuracy'], label='val acc')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-30T08:18:02.406435Z","iopub.execute_input":"2022-09-30T08:18:02.407322Z","iopub.status.idle":"2022-09-30T08:18:02.829673Z","shell.execute_reply.started":"2022-09-30T08:18:02.407253Z","shell.execute_reply":"2022-09-30T08:18:02.828723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"complete_datagen = ImageDataGenerator(rescale=1./255)\ncomplete_generator = complete_datagen.flow_from_dataframe(dataframe=train_df,\n                                                          x_col=\"fullpath\",\n                                                          target_size=(299,299),\n                                                          batch_size=1,\n                                                          shuffle=False,\n                                                          class_mode=None)\n\nSTEP_SIZE_COMPLETE = complete_generator.n//complete_generator.batch_size\ntrain_preds = iv3.predict_generator(complete_generator, steps=STEP_SIZE_COMPLETE,verbose = 1)\ntrain_preds = [np.argmax(pred) for pred in train_preds]","metadata":{"execution":{"iopub.status.busy":"2022-09-30T08:18:09.002729Z","iopub.execute_input":"2022-09-30T08:18:09.003088Z","iopub.status.idle":"2022-09-30T08:19:17.752179Z","shell.execute_reply.started":"2022-09-30T08:18:09.003056Z","shell.execute_reply":"2022-09-30T08:19:17.751237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, cohen_kappa_score,accuracy_score, f1_score\nprint(\"Train Cohen Kappa score: %.3f\" % cohen_kappa_score(train_preds, train_df['diagnosis'].astype('int'), weights='quadratic'))\nprint(\"Train Accuracy score : %.3f\" % accuracy_score(train_df['diagnosis'].astype('int'),train_preds))","metadata":{"execution":{"iopub.status.busy":"2022-09-30T08:19:17.755643Z","iopub.execute_input":"2022-09-30T08:19:17.755925Z","iopub.status.idle":"2022-09-30T08:19:17.768038Z","shell.execute_reply.started":"2022-09-30T08:19:17.755899Z","shell.execute_reply":"2022-09-30T08:19:17.766925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train F1-score : %.3f\" % f1_score(train_df['diagnosis'].astype('int'),train_preds, average='weighted'))","metadata":{"execution":{"iopub.status.busy":"2022-09-30T08:19:17.771381Z","iopub.execute_input":"2022-09-30T08:19:17.771641Z","iopub.status.idle":"2022-09-30T08:19:17.782883Z","shell.execute_reply.started":"2022-09-30T08:19:17.771617Z","shell.execute_reply":"2022-09-30T08:19:17.781960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}