{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import json\nimport math\nimport os\n\nimport cv2\nfrom PIL import Image\nimport numpy as np\nfrom keras import layers\nfrom keras.applications import DenseNet121\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nimport scipy\nimport tensorflow as tf\nfrom tqdm import tqdm\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.random.seed(2019)\ntf.set_random_seed(2019)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('../input/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nsubmission = pd.read_csv('../input/aptos2019-blindness-detection/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = '../input/aptos2019-blindness-detection/train_images/'\ntest_path = '../input/aptos2019-blindness-detection/test_images/'\n\ntrain_ids = train['id_code'].values\ntest_ids = test['id_code'].values\n\ntrain_paths = []\nfor train_id in train_ids:\n    image = train_id + '.png'\n    path = os.path.join(train_path,image)\n    train_paths.append(path)\n    \ntrain_paths = np.array(train_paths)\ntrain['path'] = train_paths","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_paths = []\nfor test_id in test_ids:\n    image = test_id + '.png'\n    path = os.path.join(test_path,image)\n    test_paths.append(path)\n    \ntest_paths = np.array(test_paths)\ntest['path'] = test_paths","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def find_radius(mid_pixels,mid_y_pixels,threshold_x,threshold_y):\n    \n    start_x = 0\n    end_x = mid_pixels.shape[0] - 1\n    \n    start_y = 0\n    end_y = mid_y_pixels.shape[0] - 1\n    \n    while True:\n        if np.sum(mid_pixels[start_x,:])>threshold_x:\n            break\n        start_x +=1\n    while True:\n        if np.sum(mid_pixels[end_x,:])>threshold_x:\n            break\n        end_x -= 1\n        \n    while True:\n        if np.sum(mid_y_pixels[start_y,:])>threshold_y:\n            break\n        start_y +=1\n    while True:\n        if np.sum(mid_y_pixels[end_y,:])>threshold_y:\n            break\n        end_y -= 1\n        \n    return start_x,end_x,start_y,end_y\n    \n    \n    \ndef preprocess_image(img):\n    mid = img.shape[1]//2\n    mid_pixels = img[mid,:]\n    mid_y_pixels = img[:,mid]\n    threshold_x = np.mean(mid_pixels)\n    threshold_y = np.mean(mid_y_pixels)\n    startx,endx,starty,endy = find_radius(mid_pixels,mid_y_pixels,threshold_x,threshold_y)\n    return cv2.resize(img[starty:endy,startx:endx],(img.shape[0],img.shape[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_df,validation_df = train_test_split(train,test_size = 0.2,stratify=train['diagnosis'].values,random_state = 42)\nprint (len(train_df),len(validation_df))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from collections import Counter\ndef get_class_weights(y):\n    counter = Counter(y)\n    majority = max(counter.values())\n    return  {cls: round(float(majority)/float(count), 2) for cls, count in counter.items()}\n\nclass_weights = get_class_weights(train_df['diagnosis'].values)\nclass_weights","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen = ImageDataGenerator(\n        zoom_range=0.2,\n        rescale = 1./255,\n        fill_mode = 'constant',\n        horizontal_flip = True,\n        vertical_flip = True,\n        preprocessing_function = preprocess_image\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['id'] = train_df['id_code'].apply(lambda x: str(x)+'.png')\ntrain_df['diagnosis'] = train_df['diagnosis'].apply(lambda x:str(x))\n\ntrain_generator = datagen.flow_from_dataframe(\ndataframe=train_df,\ndirectory=\"../input/aptos2019-blindness-detection/train_images/\",\nx_col=\"id\",\ny_col=\"diagnosis\",\nbatch_size=32,\nseed=42,\nshuffle=True,\nclass_mode=\"categorical\",\ncolor_mode = 'rgb',\ntarget_size=(224,224))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"validation_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"validation_x = []\nfor path in tqdm(validation_df['path'].values):\n    img = cv2.resize(cv2.imread(path),(224,224))\n    img = cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    validation_x.append(img)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"validation_x = np.array(validation_x)\nvalidation_x = validation_x.astype(np.float32)/255.0\nprint (validation_x.shape)\nprint (np.amin(validation_x),np.amax(validation_x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.utils import to_categorical\nvalidation_y = to_categorical(validation_df['diagnosis'].values,5)\nprint (validation_y.shape)\nprint (validation_y[:5])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Model: DenseNet-121"},{"metadata":{"trusted":true},"cell_type":"code","source":"densenet = DenseNet121(\n    weights='../input/densenet-keras/DenseNet-BC-121-32-no-top.h5',\n    include_top=False,\n    input_shape=(224,224,3)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(densenet)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(5, activation='softmax'))\n    \n    model.compile(\n        loss='categorical_crossentropy',\n        optimizer=Adam(lr=0.00005),\n        metrics=['accuracy']\n    )\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = build_model()\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training & Evaluation"},{"metadata":{"trusted":true},"cell_type":"code","source":"class Metrics(Callback):\n    def on_train_begin(self, logs={}):\n        self.val_kappas = []\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, y_val = self.validation_data[0],self.validation_data[1]\n                \n        y_val = np.argmax(y_val,axis=1)\n        \n        y_pred = self.model.predict(X_val)\n        y_pred = np.argmax(y_pred,axis=1)\n\n        _val_kappa = cohen_kappa_score(\n            y_val,\n            y_pred, \n            weights='quadratic'\n        )\n\n        self.val_kappas.append(_val_kappa)\n\n        print(f\"val_kappa: {_val_kappa:.4f}\")\n        \n        if _val_kappa == max(self.val_kappas):\n            print(\"Validation Kappa has improved. Saving model.\")\n            self.model.save('model.h5')\n\n        return","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kappa_metrics = Metrics()\n\nhistory = model.fit_generator(train_generator,validation_data = (validation_x,validation_y),\n                              epochs = 10,steps_per_epoch = len(train_df)/32,callbacks = [kappa_metrics],verbose=1,\n                              class_weight = class_weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['acc', 'val_acc']].plot()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(kappa_metrics.val_kappas)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.models import load_model\nmodel = load_model('model.h5')\nmodel.evaluate(validation_x,validation_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ntest['id'] = test['id_code'].apply(lambda x: str(x)+'.png')\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ntest_datagen = ImageDataGenerator(rescale=1./255,horizontal_flip=True,vertical_flip=True)\n\ntest_generator=test_datagen.flow_from_dataframe(dataframe=test,\n                                                directory = test_path,\n                                                x_col=\"id\",\n                                                target_size=(224,224),\n                                                batch_size=1,\n                                                shuffle=False, \n                                                class_mode=None, \n                                                seed=42)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tta_steps = 5\npreds_tta=[]\nfor i in tqdm(range(tta_steps)):\n    test_generator.reset()\n    preds = model.predict_generator(test_generator,steps = test.shape[0])\n    preds_tta.append(preds)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"final_pred = np.mean(preds_tta, axis=0)\npredicted_class_indices = np.argmax(final_pred, axis=1)\nCounter(predicted_class_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['diagnosis'] = predicted_class_indices\nsubmission.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"}},"nbformat":4,"nbformat_minor":1}