{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import confusion_matrix\nfrom mlxtend.plotting import plot_confusion_matrix\n\nfrom keras import models\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.optimizers import RMSprop,Adam\nfrom keras.utils import to_categorical\nfrom PIL import Image\nimport tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\ncwd=os.getcwd()\nos.chdir(cwd)\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path_leaf = []\ntrain_path_leaf=\"/kaggle/input/cassava-leaf-disease-classification/train_images/\"\nfor path in os.listdir(train_path_leaf):\n    if '.jpg' in path:\n        path_leaf.append(os.path.join(train_path_leaf,path))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path='/kaggle/input/cassava-leaf-disease-classification/'\ndata=pd.read_csv(path+'train.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimages=[np.array((Image.open(path)).resize((128,128))) for path in path_leaf]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images = np.array(images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nfig,ax=plt.subplots(2,5)\nfig.suptitle(\"Real Images\")\nidx=800\n\nfor i in range(0,2):\n    for j in range(0,5):\n        ax[i,j].imshow(images[idx].reshape(128,128,3))\n        idx+=900\n        \nplt.tight_layout()\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['label'].value_counts().plot(kind='bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_label = np.array(list(map(int, data['label'])))\nimage_label=to_categorical(image_label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import json\nwork_dir = '../input/cassava-leaf-disease-classification/'\nfile = open(work_dir + 'label_num_to_disease_map.json')\nlabels_leaf = json.load(file)\n\nlabels_leaf = {int(k):v for k,v in labels_leaf.items()}\nlabels_leaf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(images, image_label, test_size=0.2, random_state=42, stratify= image_label, shuffle=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 25\nbatch_size = 16","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.models import Sequential\n\nmodel2 = Sequential()\n\nmodel2.add(Conv2D(filters = 32, kernel_size = (3,3),padding = 'Same', activation ='relu', input_shape = (128,128,3)))\nmodel2.add(BatchNormalization())\nmodel2.add(Conv2D(filters = 32, kernel_size = (3,3), activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=(2,2),padding = 'Same'))\nmodel2.add(Dropout(0.25))\n\nmodel2.add(Conv2D(filters = 64, kernel_size = (3,3),padding = 'Same', activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(Conv2D(filters = 64, kernel_size = (3,3), activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=(2,2), strides=(2,2)))\nmodel2.add(Dropout(0.25))\n\nmodel2.add(Conv2D(filters = 128, kernel_size = (5,5),padding = 'Same', activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(Conv2D(filters = 128, kernel_size = (5,5), activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=(2,2), strides=(2,2)))\nmodel2.add(Dropout(0.25))\n\nmodel2.add(Conv2D(filters = 256, kernel_size = (5,5),padding = 'Same', activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(Conv2D(filters = 256, kernel_size = (5,5), activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=(2,2), strides=(2,2)))\nmodel2.add(Dropout(0.25))\n\nmodel2.add(Conv2D(filters = 512, kernel_size = (7,7),padding = 'Same', activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(Conv2D(filters = 1024, kernel_size = (7,7),padding = 'Same', activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=(2,2), strides=(2,2)))\nmodel2.add(Dropout(0.2))\n\nmodel2.add(Conv2D(filters = 2048, kernel_size = (5,5),padding = 'Same', activation ='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=(2,2), strides=(2,2)))\nmodel2.add(Dropout(0.2))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2.add(Flatten())\nmodel2.add(Dense(512, activation = \"relu\"))\nmodel2.add(BatchNormalization())\nmodel2.add(Dropout(0.3))\n\nmodel2.add(Dense(5, activation = \"softmax\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.optimizers import RMSprop,Adam,Adagrad,Adamax,Adadelta\nfrom keras.callbacks import ReduceLROnPlateau, EarlyStopping\noptimizer = Adamax()\nmodel2.compile(optimizer=optimizer, loss = 'categorical_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nlr_reduction = ReduceLROnPlateau(monitor='val_acc', \n                                            patience=3, \n                                            verbose=1, \n                                            factor=0.5, \n                                            min_lr=0.00001)\nes = EarlyStopping(monitor='val_loss',min_delta=0.0000,patience=5,verbose=0, mode='auto')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\ndatagen = ImageDataGenerator(\n        featurewise_center=False,  # set input mean to 0 over the dataset\n        samplewise_center=False,  # set each sample mean to 0\n        featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        samplewise_std_normalization=False,  # divide each input by its std\n        zca_whitening=False,  # apply ZCA whitening\n        rotation_range=10,  # randomly rotate images in the range (degrees, 0 to 180)\n        zoom_range = 0.1, # Randomly zoom image \n        width_shift_range=0.12,  # randomly shift images horizontally (fraction of total width)\n        height_shift_range=0.12,  # randomly shift images vertically (fraction of total height)\n        horizontal_flip=True,  # randomly flip images\n        vertical_flip=True)  # randomly flip images\ndatagen.fit(X_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history2 = model2.fit_generator(datagen.flow(X_train,y_train, batch_size=batch_size),\n                              epochs = epochs, validation_data = (X_test,y_test), steps_per_epoch=X_train.shape[0] // batch_size\n                              , callbacks=[lr_reduction], shuffle=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_leaf = []\ntest_path_leaf=\"/kaggle/input/cassava-leaf-disease-classification/test_images/\"\nfor path in os.listdir(test_path_leaf):\n    if '.jpg' in path:\n        test_leaf.append(os.path.join(test_path_leaf,path))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images=[np.array((Image.open(path)).resize((128,128))) for path in test_leaf]\n\ntest_images = np.array(test_images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_image_id = []\nfor path in os.listdir(test_path_leaf):\n    if '.jpg' in path:\n        test_image_id.append(path)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# predict results\nresults = model2.predict(test_images)\n\n# select the indix with the maximum probability\nresults = np.argmax(results,axis = 1)\n\nresults = pd.Series(results,name=\"label\",dtype=int)\n\nsubmission = pd.concat([pd.Series(test_image_id,name = \"image_id\"),results],axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"OUTPUT_DIR = './'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv(OUTPUT_DIR +\"submission.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}