{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import json\nimport math\nimport os\n\nimport cv2\nfrom PIL import Image\nimport numpy as np\nfrom keras import layers\nfrom keras.applications import DenseNet121\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nimport scipy\nimport tensorflow as tf\nfrom keras.callbacks import EarlyStopping,ReduceLROnPlateau,LearningRateScheduler\nfrom tqdm import tqdm_notebook as tqdm\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:36:19.471477Z","iopub.execute_input":"2021-11-01T16:36:19.471852Z","iopub.status.idle":"2021-11-01T16:36:24.488511Z","shell.execute_reply.started":"2021-11-01T16:36:19.471775Z","shell.execute_reply":"2021-11-01T16:36:24.487599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport timeit\n\ndevice_name = tf.test.gpu_device_name()\nprint(device_name)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:36:33.354430Z","iopub.execute_input":"2021-11-01T16:36:33.354751Z","iopub.status.idle":"2021-11-01T16:36:34.972516Z","shell.execute_reply.started":"2021-11-01T16:36:33.354724Z","shell.execute_reply":"2021-11-01T16:36:34.971562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(2020)\ntf.random.set_seed(2020)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:47:53.857875Z","iopub.execute_input":"2021-11-01T16:47:53.858280Z","iopub.status.idle":"2021-11-01T16:47:53.862687Z","shell.execute_reply.started":"2021-11-01T16:47:53.858245Z","shell.execute_reply":"2021-11-01T16:47:53.861659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntest_df = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nprint(train_df.shape)\nprint(test_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:47:58.153396Z","iopub.execute_input":"2021-11-01T16:47:58.153708Z","iopub.status.idle":"2021-11-01T16:47:58.201019Z","shell.execute_reply.started":"2021-11-01T16:47:58.153679Z","shell.execute_reply":"2021-11-01T16:47:58.200053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['diagnosis'].hist()\ntrain_df['diagnosis'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:48:02.198439Z","iopub.execute_input":"2021-11-01T16:48:02.198773Z","iopub.status.idle":"2021-11-01T16:48:02.387998Z","shell.execute_reply.started":"2021-11-01T16:48:02.198745Z","shell.execute_reply":"2021-11-01T16:48:02.387202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_samples(df, columns=4, rows=3):\n    fig=plt.figure(figsize=(5*columns, 4*rows))\n\n    for i in range(columns*rows):\n        image_path = df.loc[i,'id_code']\n        image_id = df.loc[i,'diagnosis']\n        img = cv2.imread(f'../input/aptos2019-blindness-detection/train_images/{image_path}.png')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        fig.add_subplot(rows, columns, i+1)\n        plt.title(image_id)\n        plt.imshow(img)\n    \n    plt.tight_layout()\n\ndisplay_samples(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:48:06.427597Z","iopub.execute_input":"2021-11-01T16:48:06.427977Z","iopub.status.idle":"2021-11-01T16:48:15.535647Z","shell.execute_reply.started":"2021-11-01T16:48:06.427942Z","shell.execute_reply":"2021-11-01T16:48:15.534419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ht = 256\nimg_wd = 256\n","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:55:03.559233Z","iopub.execute_input":"2021-11-01T16:55:03.559558Z","iopub.status.idle":"2021-11-01T16:55:03.564018Z","shell.execute_reply.started":"2021-11-01T16:55:03.559532Z","shell.execute_reply":"2021-11-01T16:55:03.562586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(image_path, desired_size=256):\n    im = Image.open(image_path)\n    im = im.resize((desired_size, )*2, resample=Image.LANCZOS)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:56:21.219461Z","iopub.execute_input":"2021-11-01T16:56:21.219806Z","iopub.status.idle":"2021-11-01T16:56:21.224392Z","shell.execute_reply.started":"2021-11-01T16:56:21.219778Z","shell.execute_reply":"2021-11-01T16:56:21.223301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = train_df.shape[0]\nx_train = np.empty((N, img_ht, img_wd, 3), dtype=np.uint8)\n\nwith tf.device('/gpu:0'):\n    for i, image_id in enumerate(tqdm(train_df['id_code'])):\n        x_train[i, :, :, :] = preprocess_image(\n\n            f'../input/aptos2019-blindness-detection/train_images/{image_id}.png'\n        )","metadata":{"execution":{"iopub.status.busy":"2021-11-01T16:56:25.618715Z","iopub.execute_input":"2021-11-01T16:56:25.619024Z","iopub.status.idle":"2021-11-01T17:08:26.894052Z","shell.execute_reply.started":"2021-11-01T16:56:25.618996Z","shell.execute_reply":"2021-11-01T17:08:26.893201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(train_df['diagnosis']).values\n\nprint(x_train.shape)\nprint(y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:08:34.618498Z","iopub.execute_input":"2021-11-01T17:08:34.618809Z","iopub.status.idle":"2021-11-01T17:08:34.629109Z","shell.execute_reply.started":"2021-11-01T17:08:34.618781Z","shell.execute_reply":"2021-11-01T17:08:34.628081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_multi = np.empty(y_train.shape, dtype=y_train.dtype)\ny_train_multi[:, 4] = y_train[:, 4]\n\nfor i in range(3, -1, -1):\n    y_train_multi[:, i] = np.logical_or(y_train[:, i], y_train_multi[:, i+1])\n\nprint(\"Original y_train:\", y_train.sum(axis=0))\nprint(\"Multilabel version:\", y_train_multi.sum(axis=0))","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:08:40.454252Z","iopub.execute_input":"2021-11-01T17:08:40.454586Z","iopub.status.idle":"2021-11-01T17:08:40.462444Z","shell.execute_reply.started":"2021-11-01T17:08:40.454556Z","shell.execute_reply":"2021-11-01T17:08:40.461226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(\n    x_train, y_train_multi, \n    test_size=0.15, \n    random_state=2019\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:08:44.916809Z","iopub.execute_input":"2021-11-01T17:08:44.917140Z","iopub.status.idle":"2021-11-01T17:08:45.730548Z","shell.execute_reply.started":"2021-11-01T17:08:44.917109Z","shell.execute_reply":"2021-11-01T17:08:45.729599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 4\n\ndef create_datagen():\n    return ImageDataGenerator(\n        zoom_range=0.15,  # set range for random zoom\n        # set mode for filling points outside the input boundaries\n        fill_mode='constant',\n        cval=0.,  # value used for fill_mode = \"constant\"\n        horizontal_flip=True,  # randomly flip images\n        vertical_flip=True,  # randomly flip images\n        width_shift_range = 0.3,\n        height_shift_range=0.3\n    )\n\n# Using original generator\ndata_generator = create_datagen().flow(x_train, y_train, batch_size=BATCH_SIZE, seed=2020)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:08:50.933538Z","iopub.execute_input":"2021-11-01T17:08:50.933925Z","iopub.status.idle":"2021-11-01T17:08:53.086544Z","shell.execute_reply.started":"2021-11-01T17:08:50.933868Z","shell.execute_reply":"2021-11-01T17:08:53.085452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(effnet)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(5, activation='sigmoid'))\n    \n    model.compile(\n        loss='binary_crossentropy',\n        optimizer=Adam(lr=0.00005),\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:08:59.772265Z","iopub.execute_input":"2021-11-01T17:08:59.772593Z","iopub.status.idle":"2021-11-01T17:08:59.777684Z","shell.execute_reply.started":"2021-11-01T17:08:59.772566Z","shell.execute_reply":"2021-11-01T17:08:59.776666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet==1.1.0","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:03.720015Z","iopub.execute_input":"2021-11-01T17:09:03.720445Z","iopub.status.idle":"2021-11-01T17:09:11.700295Z","shell.execute_reply.started":"2021-11-01T17:09:03.720402Z","shell.execute_reply":"2021-11-01T17:09:11.699026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:17.761292Z","iopub.execute_input":"2021-11-01T17:09:17.761679Z","iopub.status.idle":"2021-11-01T17:09:18.053056Z","shell.execute_reply.started":"2021-11-01T17:09:17.761642Z","shell.execute_reply":"2021-11-01T17:09:18.052242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ht = 256\nimg_wd = 256","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:22.998598Z","iopub.execute_input":"2021-11-01T17:09:22.998924Z","iopub.status.idle":"2021-11-01T17:09:23.002918Z","shell.execute_reply.started":"2021-11-01T17:09:22.998890Z","shell.execute_reply":"2021-11-01T17:09:23.001976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effnet = efn.EfficientNetB5(weights=None,\n                        include_top=False,\n                        input_shape=(img_wd, img_ht, 3))\neffnet.load_weights('../input/efficientnet/efficientnet-b5_imagenet_1000_notop.h5/efficientnet-b5_imagenet_1000_notop.h5')","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:27.315868Z","iopub.execute_input":"2021-11-01T17:09:27.316242Z","iopub.status.idle":"2021-11-01T17:09:33.766747Z","shell.execute_reply.started":"2021-11-01T17:09:27.316186Z","shell.execute_reply":"2021-11-01T17:09:33.765848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\n","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:37.887406Z","iopub.execute_input":"2021-11-01T17:09:37.887744Z","iopub.status.idle":"2021-11-01T17:09:39.321991Z","shell.execute_reply.started":"2021-11-01T17:09:37.887716Z","shell.execute_reply":"2021-11-01T17:09:39.321177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Metrics(Callback):\n    def on_train_begin(self, logs={}):\n        self.val_kappas = []\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, Y_val = x_val, y_val\n        Y_val = Y_val.sum(axis=1) - 1\n        \n        y_pred = self.model.predict(X_val) > 0.5\n        y_pred = y_pred.astype(int).sum(axis=1) - 1\n\n        _val_kappa = cohen_kappa_score(\n            Y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:45.138954Z","iopub.execute_input":"2021-11-01T17:09:45.139320Z","iopub.status.idle":"2021-11-01T17:09:45.147446Z","shell.execute_reply.started":"2021-11-01T17:09:45.139288Z","shell.execute_reply":"2021-11-01T17:09:45.146213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_loss',\n                                      mode='auto',\n                                      verbose=1,\n                                      patience=10)\n\nlearning_rate_reduction = ReduceLROnPlateau(monitor='val_loss',\n                                            patience=3,\n                                            verbose=1,\n                                            mode = 'auto',\n                                            factor=0.25,\n                                            min_lr=0.000001)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:09:50.786283Z","iopub.execute_input":"2021-11-01T17:09:50.786622Z","iopub.status.idle":"2021-11-01T17:09:50.792523Z","shell.execute_reply.started":"2021-11-01T17:09:50.786591Z","shell.execute_reply":"2021-11-01T17:09:50.791314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/gpu:0'):\n    kappa_metrics = Metrics()\n    history = model.fit_generator(\n        data_generator,\n        steps_per_epoch=x_train.shape[0] / BATCH_SIZE,\n        #steps_per_epoch=5,\n        epochs=15,\n        validation_data=(x_val, y_val),\n        callbacks=[kappa_metrics,es, learning_rate_reduction]\n    )","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-11-01T17:09:55.223810Z","iopub.execute_input":"2021-11-01T17:09:55.224146Z","iopub.status.idle":"2021-11-01T17:44:19.277407Z","shell.execute_reply.started":"2021-11-01T17:09:55.224116Z","shell.execute_reply":"2021-11-01T17:44:19.275982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('history.json', 'w') as f:\n    try:\n        json.dump(history.history, f)\n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:47:22.288708Z","iopub.execute_input":"2021-11-01T17:47:22.289028Z","iopub.status.idle":"2021-11-01T17:47:22.294291Z","shell.execute_reply.started":"2021-11-01T17:47:22.289001Z","shell.execute_reply":"2021-11-01T17:47:22.293399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.ylim((0.7,1.0))\nplt.plot(kappa_metrics.val_kappas)","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:47:30.563602Z","iopub.execute_input":"2021-11-01T17:47:30.563937Z","iopub.status.idle":"2021-11-01T17:47:30.722986Z","shell.execute_reply.started":"2021-11-01T17:47:30.563906Z","shell.execute_reply":"2021-11-01T17:47:30.722142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-02T09:08:53.468007Z","iopub.execute_input":"2021-07-02T09:08:53.468394Z","iopub.status.idle":"2021-07-02T09:08:53.580657Z","shell.execute_reply.started":"2021-07-02T09:08:53.468361Z","shell.execute_reply":"2021-07-02T09:08:53.579822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nprint(history_df.head())\n\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['accuracy', 'val_accuracy']].plot()","metadata":{"execution":{"iopub.status.busy":"2021-11-01T17:47:35.806793Z","iopub.execute_input":"2021-11-01T17:47:35.807098Z","iopub.status.idle":"2021-11-01T17:47:36.093708Z","shell.execute_reply.started":"2021-11-01T17:47:35.807070Z","shell.execute_reply":"2021-11-01T17:47:36.092907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}