{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import json\nimport math\nimport os\n\nimport cv2\nfrom PIL import Image\nimport numpy as np\nfrom keras import layers\nfrom keras.applications import DenseNet121\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nimport scipy\nimport tensorflow as tf\nfrom keras.callbacks import EarlyStopping,ReduceLROnPlateau,LearningRateScheduler\nfrom tqdm import tqdm_notebook as tqdm\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:33:50.798254Z","iopub.execute_input":"2021-07-02T07:33:50.798566Z","iopub.status.idle":"2021-07-02T07:33:51.552407Z","shell.execute_reply.started":"2021-07-02T07:33:50.798536Z","shell.execute_reply":"2021-07-02T07:33:51.551658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport timeit\n\ndevice_name = tf.test.gpu_device_name()\nprint(device_name)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:33:51.553895Z","iopub.execute_input":"2021-07-02T07:33:51.554216Z","iopub.status.idle":"2021-07-02T07:33:51.563986Z","shell.execute_reply.started":"2021-07-02T07:33:51.554181Z","shell.execute_reply":"2021-07-02T07:33:51.563063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(2020)\ntf.random.set_seed(2020)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:34:44.127393Z","iopub.execute_input":"2021-07-02T07:34:44.127800Z","iopub.status.idle":"2021-07-02T07:34:44.131602Z","shell.execute_reply.started":"2021-07-02T07:34:44.127766Z","shell.execute_reply":"2021-07-02T07:34:44.130824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntest_df = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint(train_df.shape)\nprint(test_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:34:54.978420Z","iopub.execute_input":"2021-07-02T07:34:54.978743Z","iopub.status.idle":"2021-07-02T07:34:55.023558Z","shell.execute_reply.started":"2021-07-02T07:34:54.978702Z","shell.execute_reply":"2021-07-02T07:34:55.022918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['diagnosis'].hist()\ntrain_df['diagnosis'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:34:56.386711Z","iopub.execute_input":"2021-07-02T07:34:56.387050Z","iopub.status.idle":"2021-07-02T07:34:56.572543Z","shell.execute_reply.started":"2021-07-02T07:34:56.387020Z","shell.execute_reply":"2021-07-02T07:34:56.571869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'../input/aptos2019-blindness-detection/train_images/{image_path}.png')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:34:57.378989Z","iopub.execute_input":"2021-07-02T07:34:57.379326Z","iopub.status.idle":"2021-07-02T07:35:05.700951Z","shell.execute_reply.started":"2021-07-02T07:34:57.379297Z","shell.execute_reply":"2021-07-02T07:35:05.699905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ht = 256\nimg_wd = 256\n","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:35:08.658990Z","iopub.execute_input":"2021-07-02T07:35:08.659314Z","iopub.status.idle":"2021-07-02T07:35:08.662920Z","shell.execute_reply.started":"2021-07-02T07:35:08.659284Z","shell.execute_reply":"2021-07-02T07:35:08.662032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(image_path, desired_size=256):\n    im = Image.open(image_path)\n    im = im.resize((desired_size, )*2, resample=Image.LANCZOS)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:35:09.954189Z","iopub.execute_input":"2021-07-02T07:35:09.954496Z","iopub.status.idle":"2021-07-02T07:35:09.960614Z","shell.execute_reply.started":"2021-07-02T07:35:09.954467Z","shell.execute_reply":"2021-07-02T07:35:09.959627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = train_df.shape[0]\nx_train = np.empty((N, img_ht, img_wd, 3), dtype=np.uint8)\n\nwith tf.device('/gpu:0'):\n    for i, image_id in enumerate(tqdm(train_df['id_code'])):\n        x_train[i, :, :, :] = preprocess_image(\n\n            f'../input/aptos2019-blindness-detection/train_images/{image_id}.png'\n        )","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:35:10.946848Z","iopub.execute_input":"2021-07-02T07:35:10.947167Z","iopub.status.idle":"2021-07-02T07:46:19.991580Z","shell.execute_reply.started":"2021-07-02T07:35:10.947139Z","shell.execute_reply":"2021-07-02T07:46:19.990700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['diagnosis']).values\n\nprint(x_train.shape)\nprint(y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:23.489999Z","iopub.execute_input":"2021-07-02T07:46:23.490568Z","iopub.status.idle":"2021-07-02T07:46:23.498211Z","shell.execute_reply.started":"2021-07-02T07:46:23.490522Z","shell.execute_reply":"2021-07-02T07:46:23.497337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_multi = np.empty(y_train.shape, dtype=y_train.dtype)\ny_train_multi[:, 4] = y_train[:, 4]\n\nfor i in range(3, -1, -1):\n    y_train_multi[:, i] = np.logical_or(y_train[:, i], y_train_multi[:, i+1])\n\nprint(\"Original y_train:\", y_train.sum(axis=0))\nprint(\"Multilabel version:\", y_train_multi.sum(axis=0))","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:24.448245Z","iopub.execute_input":"2021-07-02T07:46:24.448581Z","iopub.status.idle":"2021-07-02T07:46:24.457871Z","shell.execute_reply.started":"2021-07-02T07:46:24.448549Z","shell.execute_reply":"2021-07-02T07:46:24.456647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    x_train, y_train_multi, \n    test_size=0.15, \n    random_state=2019\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:25.424537Z","iopub.execute_input":"2021-07-02T07:46:25.424865Z","iopub.status.idle":"2021-07-02T07:46:25.640122Z","shell.execute_reply.started":"2021-07-02T07:46:25.424834Z","shell.execute_reply":"2021-07-02T07:46:25.639341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 4\n\ndef create_datagen():\n    return ImageDataGenerator(\n        zoom_range=0.15,  # set range for random zoom\n        # set mode for filling points outside the input boundaries\n        fill_mode='constant',\n        cval=0.,  # value used for fill_mode = \"constant\"\n        horizontal_flip=True,  # randomly flip images\n        vertical_flip=True,  # randomly flip images\n        width_shift_range = 0.3,\n        height_shift_range=0.3\n    )\n\n# Using original generator\ndata_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE, seed=2020)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:25.975578Z","iopub.execute_input":"2021-07-02T07:46:25.975923Z","iopub.status.idle":"2021-07-02T07:46:26.772001Z","shell.execute_reply.started":"2021-07-02T07:46:25.975892Z","shell.execute_reply":"2021-07-02T07:46:26.771071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(5, activation='sigmoid'))\n    \n    model.compile(\n        loss='binary_crossentropy',\n        optimizer=Adam(lr=0.00005),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:28.570054Z","iopub.execute_input":"2021-07-02T07:46:28.570472Z","iopub.status.idle":"2021-07-02T07:46:28.578506Z","shell.execute_reply.started":"2021-07-02T07:46:28.570433Z","shell.execute_reply":"2021-07-02T07:46:28.577549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet==1.1.0","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:32.600246Z","iopub.execute_input":"2021-07-02T07:46:32.600576Z","iopub.status.idle":"2021-07-02T07:46:40.426048Z","shell.execute_reply.started":"2021-07-02T07:46:32.600547Z","shell.execute_reply":"2021-07-02T07:46:40.425027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:40.428070Z","iopub.execute_input":"2021-07-02T07:46:40.428425Z","iopub.status.idle":"2021-07-02T07:46:40.731175Z","shell.execute_reply.started":"2021-07-02T07:46:40.428387Z","shell.execute_reply":"2021-07-02T07:46:40.730378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ht = 256\nimg_wd = 256","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:46:40.732332Z","iopub.execute_input":"2021-07-02T07:46:40.732659Z","iopub.status.idle":"2021-07-02T07:46:40.738569Z","shell.execute_reply.started":"2021-07-02T07:46:40.732625Z","shell.execute_reply":"2021-07-02T07:46:40.735581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effnet = efn.EfficientNetB5(weights=None,\n                        include_top=False,\n                        input_shape=(img_wd, img_ht, 3))\neffnet.load_weights('../input/efficientnet/efficientnet-b5_imagenet_1000_notop.h5/efficientnet-b5_imagenet_1000_notop.h5')","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:47:10.413353Z","iopub.execute_input":"2021-07-02T07:47:10.413683Z","iopub.status.idle":"2021-07-02T07:47:17.674230Z","shell.execute_reply.started":"2021-07-02T07:47:10.413650Z","shell.execute_reply":"2021-07-02T07:47:17.673437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:47:19.674531Z","iopub.execute_input":"2021-07-02T07:47:19.674886Z","iopub.status.idle":"2021-07-02T07:47:21.069922Z","shell.execute_reply.started":"2021-07-02T07:47:19.674853Z","shell.execute_reply":"2021-07-02T07:47:21.069157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def on_train_begin(self, logs={}):\n        self.val_kappas = []\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, Y_val = x_val, y_val\n        Y_val = Y_val.sum(axis=1) - 1\n        \n        y_pred = self.model.predict(X_val) > 0.5\n        y_pred = y_pred.astype(int).sum(axis=1) - 1\n\n        _val_kappa = cohen_kappa_score(\n            Y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:47:22.022220Z","iopub.execute_input":"2021-07-02T07:47:22.022537Z","iopub.status.idle":"2021-07-02T07:47:22.029619Z","shell.execute_reply.started":"2021-07-02T07:47:22.022506Z","shell.execute_reply":"2021-07-02T07:47:22.028655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_loss',\n                                      mode='auto',\n                                      verbose=1,\n                                      patience=10)\n\nlearning_rate_reduction = ReduceLROnPlateau(monitor='val_loss',\n                                            patience=3,\n                                            verbose=1,\n                                            mode = 'auto',\n                                            factor=0.25,\n                                            min_lr=0.000001)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:47:23.154692Z","iopub.execute_input":"2021-07-02T07:47:23.155047Z","iopub.status.idle":"2021-07-02T07:47:23.159869Z","shell.execute_reply.started":"2021-07-02T07:47:23.155014Z","shell.execute_reply":"2021-07-02T07:47:23.158854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/gpu:0'):\n    kappa_metrics = Metrics()\n    history = model.fit_generator(\n        data_generator,\n        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n        #steps_per_epoch=5,\n        epochs=15,\n        validation_data=(x_val, y_val),\n        callbacks=[kappa_metrics,es, learning_rate_reduction]\n    )","metadata":{"execution":{"iopub.status.busy":"2021-07-02T07:56:03.684301Z","iopub.execute_input":"2021-07-02T07:56:03.684655Z","iopub.status.idle":"2021-07-02T08:31:04.312219Z","shell.execute_reply.started":"2021-07-02T07:56:03.684623Z","shell.execute_reply":"2021-07-02T08:31:04.311459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('history.json', 'w') as f:\n    try:\n        json.dump(history.history, f)\n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2021-07-02T08:31:27.129002Z","iopub.execute_input":"2021-07-02T08:31:27.129349Z","iopub.status.idle":"2021-07-02T08:31:27.135723Z","shell.execute_reply.started":"2021-07-02T08:31:27.129319Z","shell.execute_reply":"2021-07-02T08:31:27.134657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.ylim((0.7,1.0))\nplt.plot(kappa_metrics.val_kappas)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T09:09:42.404164Z","iopub.execute_input":"2021-07-02T09:09:42.404621Z","iopub.status.idle":"2021-07-02T09:09:42.597238Z","shell.execute_reply.started":"2021-07-02T09:09:42.404575Z","shell.execute_reply":"2021-07-02T09:09:42.596440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-02T09:08:53.468007Z","iopub.execute_input":"2021-07-02T09:08:53.468394Z","iopub.status.idle":"2021-07-02T09:08:53.580657Z","shell.execute_reply.started":"2021-07-02T09:08:53.468361Z","shell.execute_reply":"2021-07-02T09:08:53.579822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nprint(history_df.head())\n\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['accuracy', 'val_accuracy']].plot()","metadata":{"execution":{"iopub.status.busy":"2021-07-02T09:13:12.532611Z","iopub.execute_input":"2021-07-02T09:13:12.532962Z","iopub.status.idle":"2021-07-02T09:13:12.834193Z","shell.execute_reply.started":"2021-07-02T09:13:12.532931Z","shell.execute_reply":"2021-07-02T09:13:12.833337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}