{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom PIL import Image\nimport glob\nimport gc\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport os,sys\nprint(os.listdir(\"../input\"))\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns\n%matplotlib inline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nimport itertools\nfrom keras.utils.np_utils import to_categorical # convert to one-hot-encoding\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D\nfrom keras.optimizers import RMSprop\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ReduceLROnPlateau\nfrom memory_profiler import profile\nfrom keras.layers import Conv3D, MaxPool3D\nimport random\nimport cv2\nfrom imblearn.over_sampling import SMOTE\n\nsns.set(style='white', context='notebook', palette='deep')\ngc.collect()\nprint('gc comp')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#common\n\ntrain = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nsize =256\nchannel = 3\n\nimg_name_train = train['id_code']\nimg_name_test = test['id_code']\n\n#y_train = np.asarray(train['diagnosis'])\ny_train = train['diagnosis']\n\nimg_path_train = \"../input/train_images/\"\nimg_path_test = \"../input/test_images/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#common\ng = sns.countplot(y_train)\ny_train.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for decrease samples#\n\nm = 0\n# if diagnosis = 0 decrease the smple to 1/6,if diagnosis = 2 decrease the smple to 1/3\nfor id in img_name_train:\n    df_comp = train[train['id_code']==id]\n    val_comp = df_comp['diagnosis'].item()\n    \n    if val_comp == 0:\n        \n        if random.random() > 0.166:\n            train = train.drop(m)\n    \n    elif val_comp == 2:\n        if random.random() < 0.666:\n            train = train.drop(m)\n        \n    m = m+1\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for decrease samples#\ny_train = train['diagnosis']\ng = sns.countplot(y_train)\ny_train.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for decrease samples#\nn = d0 = d1 = d2 = d3 = d4 = 0\n\ncount_0 = np.count_nonzero(y_train == 0, axis=0)\ncount_1 = np.count_nonzero(y_train == 1, axis=0)\ncount_2 = np.count_nonzero(y_train == 2, axis=0)\ncount_3 = np.count_nonzero(y_train == 3, axis=0)\ncount_4 = np.count_nonzero(y_train == 4, axis=0)\n\nx_train_0 = np.empty((count_0, size, size, channel), dtype=np.float32)\ny_train_0 = np.zeros((count_0,2), dtype=np.int)\nx_train_1 = np.empty((count_1, size, size, channel), dtype=np.float32)\ny_train_1 = np.zeros((count_1,2), dtype=np.int)\nx_train_2 = np.empty((count_2, size, size, channel), dtype=np.float32)\ny_train_2 = np.zeros((count_2,2), dtype=np.int)\nx_train_3 = np.empty((count_3, size, size, channel), dtype=np.float32)\ny_train_3 = np.zeros((count_3,2), dtype=np.int)\nx_train_4 = np.empty((count_4, size, size, channel), dtype=np.float32)\ny_train_4 = np.zeros((count_4,2), dtype=np.int)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for decrease samples#\nimg_name_train = train['id_code']\nimg_name_test = test['id_code']\n\ndef img_read(img_path,id):\n    \n    #when using cv2 and grey scale images\n    if channel == 1:\n        img = cv2.imread(img_path + id + '.png')\n        grey_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        img_rs = np.asarray(cv2.resize(grey_img,(size,size)))\n        img_rs = img_rs/255\n        img_rs = img_rs.reshape(size,size,channel)\n    \n        return img_rs\n    \n    #when using rgb images\n    elif channel == 3:\n        #img = Image.open(img_path + id + '.png')\n        img = cv2.imread(img_path + id + '.png')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img_rs = np.asarray(cv2.resize(img,(size,size)))\n        img_rs = img_rs/255\n        #img_rs = img_rs.reshape(size,size,channel)\n        \n        return img_rs\n\nfor id in img_name_train:\n    df_comp = train[train['id_code']==id]\n    val_comp = df_comp['diagnosis'].item()\n    val = np.asarray(df_comp['diagnosis'])\n    \n    if val_comp==0:\n\n        img_rs = img_read(img_path_train,id)\n        \n        y_train_0[d0,:] = np.array([val_comp])\n        x_train_0[d0,:,:,:] = img_rs\n        d0 = d0 + 1 \n        \n    elif val_comp==1:\n        \n        img_rs = img_read(img_path_train,id)\n        \n        y_train_1[d1,:] = np.array([val_comp])\n        x_train_1[d1,:,:,:] = img_rs\n        d1 = d1 + 1\n        \n    elif val_comp==2:\n        \n        img_rs = img_read(img_path_train,id)\n        \n        y_train_2[d2,:] = np.array([val_comp])\n        x_train_2[d2,:,:,:] = img_rs\n        d2 = d2 + 1\n        \n    elif val_comp==3:\n        \n        img_rs = img_read(img_path_train,id)\n        \n        y_train_3[d3,:] = np.array([val_comp])\n        x_train_3[d3,:,:,:] = img_rs\n        \n        d3 = d3 + 1\n        \n    elif val_comp==4:\n        \n        img_rs = img_read(img_path_train,id)\n        \n        y_train_4[d4,:] = np.array([val_comp])\n        x_train_4[d4,:,:,:] = img_rs\n        \n        d4 = d4 + 1\n    n = n + 1\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(y_train_0.shape)\nprint(y_train_0.shape)\nprint(d0)\nprint(n)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"##non decrease samples#\n\n#train = pd.read_csv(\"../input/train.csv\")\n#test = pd.read_csv(\"../input/test.csv\")\n#size =256\n\n#img_name_train = train['id_code']\n#img_name_test = test['id_code']\n#\n#N_train = train.shape[0]\n#N_test = test.shape[0]\n#\n#x_train = np.empty((N_train, size, size, 1), dtype=np.float32)\n##x_train = np.empty((N_train, size, size, 3), dtype=np.uint8)\n#\n#x_test = np.empty((N_test, size, size, 1), dtype=np.float32)\n##x_test = np.empty((N_test, size, size, 3), dtype=np.uint8)\n#\n#y_train = np.asarray(train['diagnosis'])\n#\n#\n#img_path_train = \"../input/train_images/\"\n#img_path_test = \"../input/test_images/\"\n#\n##train\n#n = 0\n#for id in img_name_train:\n#    #img = Image.open(img_path_train + id + '.png')\n#    #img_rs = np.asarray(img.resize((size,size)))\n#    \n#    img = cv2.imread(img_path_train + id + '.png')\n#    grey_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n#    img_rs = np.asarray(cv2.resize(grey_img,(256,256)))\n#    img_rs = img_rs/255\n#    \n#    x_train[n, :, :,:] = img_rs.reshape(256,256,1)\n#\n#    n=n+1\n#    \n##    g = plt.imshow(x_train[3][:,:,0])\n#\n###test\n##n = 0\n##for id in img_name_test:\n##    img = Image.open(img_path_test + id + '.png')\n##    img_rs = np.asarray(img.resize((size,size)))\n##    img_rs = img_rs/255\n#    \n##    x_test[n, :, :, :] = img_rs\n##    #x_test[:, :, :] = img_rs\n##    n = n + 1\n#\n#\n#print(x_train.shape)\n#print(y_train.shape)\n##print(x_test.shape)\n#print(x_train.max())\n#print(y_train.max())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#print(x_train)\n#print(y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for decrease samples#\nx_train = np.vstack([x_train_0,x_train_1,x_train_2,x_train_3,x_train_4])\ny_train = np.vstack([y_train_0,y_train_1,y_train_2,y_train_3,y_train_4])\ny_train = np.delete(y_train,0,axis=1)\nprint(x_train.shape)\nprint(y_train.shape)\nprint(y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Encode labels to one hot vectors (ex : 2 -> [0,0,1,0,0])\n\ny_train_c = to_categorical(y_train, num_classes = 5)\nprint(y_train_c.shape)\n# Set the random seed\nrandom_seed = 2\n\n# Split the train and the validation set for the fitting\nX_train, X_val, Y_train, Y_val = train_test_split(x_train, y_train_c, test_size = 0.1, random_state=random_seed)\nprint(X_train.shape)\nprint(X_val.shape)\nprint(Y_train.shape)\nprint(Y_val.shape)\n\nprint(Y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Some examples\n#X_train_show = np.asarray(img.resize((255,255)))\n#print(X_train_show.shape)\n#g = plt.imshow(X_train_show)\n#print(X_train)\n#print(X_val)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(X_train[0].shape)\nprint(X_train.max())\nprint(X_train[0].max())\n#temp = X_train[1][:,:,0].reshape(256,256)\ntemp = X_val[1][:,:,0]\n\ntemp1 = X_train[1][:,0,:]\ntemp2 = X_train[1][0,:,:]\ntemp3 = X_train[1][:,:,:]\nprint(temp.max())\nprint(temp.shape)\nprint(temp1.max())\nprint(temp1.shape)\nprint(temp2.max())\nprint(temp2.shape)\nprint(temp3.max())\nprint(temp3.shape)\n#for m in range(0,3295):\n#    print(X_train[0].max())\ng = plt.imshow(temp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Define the optimizer\noptimizer = RMSprop(lr=0.001, rho=0.9, epsilon=1e-08, decay=0.0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Set the CNN model \n# my CNN architechture is In -> [[Conv2D->relu]*2 -> MaxPool2D -> Dropout]*2 -> Flatten -> Dense -> Dropout -> Out\n\nmodel = Sequential()\n\nmodel.add(Conv2D(filters = 32 , kernel_size = (5,5),padding = 'Same',activation ='relu', input_shape = (256,256,channel)))\nmodel.add(Conv2D(filters = 32, kernel_size = (5,5),padding = 'Same',activation ='relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = (5,5),padding = 'Same',activation ='relu'))\nmodel.add(Conv2D(filters = 32, kernel_size = (5,5),padding = 'Same',activation ='relu'))\nmodel.add(MaxPool2D(pool_size=(2,2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same',activation ='relu'))\nmodel.add(MaxPool2D(pool_size=(2,2), strides=(2,2)))\nmodel.add(Dropout(0.25))\n\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = \"relu\"))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(5, activation = \"softmax\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer = optimizer , loss = \"categorical_crossentropy\", metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Set a learning rate annealer\nlearning_rate_reduction = ReduceLROnPlateau(monitor='val_acc',patience=3,verbose=1,factor=0.5,min_lr=0.00001)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 30 # Turn epochs to 30 \nbatch_size = 86\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# With data augmentation to prevent overfitting\ndatagen = ImageDataGenerator(featurewise_center=False,\n                             samplewise_center=False,\n                             featurewise_std_normalization=False,\n                             samplewise_std_normalization=False,\n                             zca_whitening=False,\n                             rotation_range=10,\n                             zoom_range= 0.1,\n                             width_shift_range=0.1,\n                             height_shift_range=0.1,\n                             horizontal_flip=False,\n                             vertical_flip=False)\ndatagen.fit(X_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Fit the model\nhistory = model.fit_generator(datagen.flow(X_train,Y_train, batch_size=batch_size),epochs = epochs, validation_data = (X_val,Y_val),verbose = 2, steps_per_epoch=X_train.shape[0] // batch_size, callbacks=[learning_rate_reduction])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot the loss and accuracy curves for training and validation \nfig, ax = plt.subplots(2,1)\nax[0].plot(history.history['loss'], color='b', label=\"Training loss\")\nax[0].plot(history.history['val_loss'], color='r', label=\"validation loss\",axes =ax[0])\nlegend = ax[0].legend(loc='best', shadow=True)\n\nax[1].plot(history.history['acc'], color='b', label=\"Training accuracy\")\nax[1].plot(history.history['val_acc'], color='r',label=\"Validation accuracy\")\nlegend = ax[1].legend(loc='best', shadow=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Look at confusion matrix \n\ndef plot_confusion_matrix(cm, classes,normalize=False,title='Confusion matrix',cmap=plt.cm.Blues):\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],horizontalalignment=\"center\",color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n\n# Predict the values from the validation dataset\nY_pred = model.predict(X_val)\n# Convert predictions classes to one hot vectors \nY_pred_classes = np.argmax(Y_pred,axis = 1) \n# Convert validation observations to one hot vectors\nY_true = np.argmax(Y_val,axis = 1) \n# compute the confusion matrix\nconfusion_mtx = confusion_matrix(Y_true, Y_pred_classes) \n# plot the confusion matrix\nplot_confusion_matrix(confusion_mtx, classes = range(5)) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Display some error results \n\n# Errors are difference between predicted labels and true labels\nerrors = (Y_pred_classes - Y_true != 0)\n\nY_pred_classes_errors = Y_pred_classes[errors]\nY_pred_errors = Y_pred[errors]\nY_true_errors = Y_true[errors]\nX_val_errors = X_val[errors]\n\ndef display_errors(errors_index,img_errors,pred_errors, obs_errors):\n    \"\"\" This function shows 6 images with their predicted and real labels\"\"\"\n    n = 0\n    nrows = 2\n    ncols = 3\n    fig, ax = plt.subplots(nrows,ncols,sharex=True,sharey=True)\n    for row in range(nrows):\n        for col in range(ncols):\n            error = errors_index[n]\n            ax[row,col].imshow((img_errors[error]).reshape((size,size,channel)))\n            ax[row,col].set_title(\"Predicted label :{}\\nTrue label :{}\".format(pred_errors[error],obs_errors[error]))\n            n += 1\n\n# Probabilities of the wrong predicted numbers\nY_pred_errors_prob = np.max(Y_pred_errors,axis = 1)\n\n# Predicted probabilities of the true values in the error set\ntrue_prob_errors = np.diagonal(np.take(Y_pred_errors, Y_true_errors, axis=1))\n\n# Difference between the probability of the predicted label and the true label\ndelta_pred_true_errors = Y_pred_errors_prob - true_prob_errors\n\n# Sorted list of the delta prob errors\nsorted_dela_errors = np.argsort(delta_pred_true_errors)\n\n# Top 6 errors \nmost_important_errors = sorted_dela_errors[-6:]\n\n# Show the top 6 errors\ndisplay_errors(most_important_errors, X_val_errors, Y_pred_classes_errors, Y_true_errors)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n = 0\ncount = np.count_nonzero(img_name_test, axis=0)\nx_test = np.empty((count, size, size, channel), dtype=np.float32)\n\nfor id in img_name_test:\n    #df_comp = test[test['id_code']==id]    \n    img_rs = img_read(img_path_test,id)\n    \n    #y_test[n,:] = np.array([val_comp])\n    x_test[n,:,:,:] = img_rs\n        \n    n = n + 1\n\nprint(n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"result = model.predict(x_test)\nresult = np.argmax(result,axis=1) \nprint(result.shape)\nprint(result)\nsub_data=pd.DataFrame({'id_code' : img_name_test, 'diagnosis' : result})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"g = sns.countplot(sub_data['diagnosis'])\nsub_data['diagnosis'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_data.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}