{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"DATA_PATH = '../input/aptos2019-blindness-detection'\n\nTRAIN_IMG_PATH = os.path.join(DATA_PATH, 'train_images')\nTEST_IMG_PATH = os.path.join(DATA_PATH, 'test_images')\nTRAIN_LABEL_PATH = os.path.join(DATA_PATH, 'train.csv')\nTEST_LABEL_PATH = os.path.join(DATA_PATH, 'test.csv')\n\ndf_train = pd.read_csv(TRAIN_LABEL_PATH)\ndf_test = pd.read_csv(TEST_LABEL_PATH)\n\nprint('num of train images ', len(os.listdir(TRAIN_IMG_PATH)))\nprint('num of test images  ', len(os.listdir(TEST_IMG_PATH)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf_train['diagnosis'] = df_train['diagnosis'].astype('str')\ndf_train = df_train[['id_code', 'diagnosis']]\nif df_train['id_code'][0].split('.')[-1] != 'png':\n    for index in range(len(df_train['id_code'])):\n        df_train['id_code'][index] = df_train['id_code'][index] + '.png'\n        \ndf_test = df_test[['id_code']]\nif df_test['id_code'][0].split('.')[-1] != 'png':\n    for index in range(len(df_test['id_code'])):\n        df_test['id_code'][index] = df_test['id_code'][index] + '.png'\n\ntrain_data = np.arange(df_train.shape[0])\ntrain_idx, val_idx = train_test_split(train_data, train_size=0.8, random_state=2019)\n\nX_train = df_train.iloc[train_idx, :]\nX_val = df_train.iloc[val_idx, :]\nX_test = df_test\n\nprint(X_train.shape)\nprint(X_val.shape)\nprint(X_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nimport keras\nnum_classes = 5\nimg_size = (224, 224, 3)\n#nb_train_samples = len(X_train)\n#nb_validation_samples = len(X_val)\nnb_test_samples = len(X_test)\n\nbatch_size = 32\n\n\ndatagen = ImageDataGenerator(\n    preprocessing_function= \\\n    keras.applications.mobilenet.preprocess_input)\ntest_batches =datagen.flow_from_dataframe(\n    dataframe=X_test,\n    directory=TEST_IMG_PATH,\n    x_col='id_code',\n    y_col=None,\n    target_size= img_size[:2],\n    color_mode='rgb',\n    class_mode=None,\n    batch_size=batch_size,\n    shuffle=False,\n    seed=2019\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"\nCreated on Tue Jul 30 21:59:33 2019\nhttps://www.kaggle.com/vbookshelf/skin-lesion-analyzer-tensorflow-js-web-app\n@author: chopinforest\n\"\"\"\n\n\nimport numpy as np\n#import keras\n#from keras import backend as K\n\nimport keras\nfrom keras.layers import Dense, Dropout\nfrom keras.optimizers import Adam\nfrom keras.metrics import categorical_crossentropy\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Model\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\n\n#image_size = 224\nnum_classes=5\n\n# create a copy of a mobilenet model\n#keras.applications.mobilenet.MobileNet(input_shape=None, alpha=1.0, depth_multiplier=1, dropout=1e-3, include_top=True, weights='imagenet', input_tensor=None, pooling=None, classes=1000)\n\nmobile = keras.applications.mobilenet.MobileNet(weights='../input/mobilenet/mobilenet_1_0_224_tf.h5')\n# CREATE THE MODEL ARCHITECTURE\n\n# Exclude the last 5 layers of the above model.\n# This will include all layers up to and including global_average_pooling2d_1\nx = mobile.layers[-6].output\n\n# Create a new dense layer for predictions\n# 7 corresponds to the number of classes\nx = Dropout(0.25)(x)\npredictions = Dense(num_classes, activation='softmax')(x)\n\n# inputs=mobile.input selects the input layer, outputs=predictions refers to the\n# dense layer we created above.\n\nmodel = Model(inputs=mobile.input, outputs=predictions)\n\n# We need to choose how many layers we actually want to be trained.\n\n# Here we are freezing the weights of all layers except the\n# last 23 layers in the new model.\n# The last 23 layers of the model will be trained.\n\nfor layer in model.layers[:-23]:\n    layer.trainable = False\n    \n    \n# Define Top2 and Top3 Accuracy\n\nfrom keras.metrics import categorical_accuracy, top_k_categorical_accuracy\n\ndef top_3_accuracy(y_true, y_pred):\n    return top_k_categorical_accuracy(y_true, y_pred, k=3)\n\ndef top_2_accuracy(y_true, y_pred):\n    return top_k_categorical_accuracy(y_true, y_pred, k=2)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_train_samples = len(X_train)\nnum_val_samples = len(X_val)\ntrain_batch_size = 32\nval_batch_size = 32\n\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)\n\n\n\n\nmodel.compile(Adam(lr=0.01), loss='categorical_crossentropy', \n              metrics=[categorical_accuracy, top_2_accuracy, top_3_accuracy])\n\n\n# Add weights to try to make the model more sensitive to melanoma\n\nclass_weights={\n    0: 1.0, # akiec\n    1: 1.0, # bcc\n    2: 1.0, # bkl\n    3: 2.0, # df\n    4: 2.0, # mel # Try to make the model more sensitive to Melanoma.\n}\n\nfilepath = \"mobilenet_224.h5\"\ncheckpoint = ModelCheckpoint(filepath, monitor='val_top_3_accuracy', verbose=1, \n                             save_best_only=True, mode='max')\n\nreduce_lr = ReduceLROnPlateau(monitor='val_top_3_accuracy', factor=0.5, patience=2, \n                                   verbose=1, mode='max', min_lr=0.00001)\n                              \n                              \ncallbacks_list = [checkpoint, reduce_lr]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from keras.models import load_model\n#from keras.models import load_weights\n\nmodel.load_weights('../input/my-new-weights/mobilenet_224.h5')\n#model=load_model('../input/my-new-weights/mobilenet_224.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tqdm import tqdm\nfrom math import ceil\n# Apply TTA\npreds_tta = []\ntta_steps = 10\nfor i in tqdm(range(tta_steps)):\n    test_batches.reset()\n    preds = model.predict_generator(\n        generator=test_batches ,\n        steps =ceil(nb_test_samples/batch_size)\n    )\n    preds_tta.append(preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds_mean = np.mean(preds_tta, axis=0)\npredicted_class_indices = np.argmax(preds_mean, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_batches = datagen.flow_from_dataframe(\n    dataframe=X_train, \n    directory=TRAIN_IMG_PATH,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=img_size[:2],\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=batch_size,\n    seed=2019\n)\nlabels = (train_batches.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\npredictions = [labels[k] for k in predicted_class_indices]\n\nsubmission = pd.read_csv(os.path.join(DATA_PATH, 'sample_submission.csv'))\nsubmission['diagnosis'] = predictions\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.pivot_table(index='diagnosis', aggfunc=len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\nplt.figure(figsize=(12, 6))\nsns.countplot(submission[\"diagnosis\"])\nplt.title(\"Number of data per each diagnosis\")\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}