{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Objectives**\n\nThe main objective of the competition is to develop machine learning-based models to accurately classify a given leaf image from the test dataset to a particular disease category, and to identify an individual disease from multiple disease symptoms on a single leaf image.","metadata":{}},{"cell_type":"markdown","source":"# **Resources**\n\nDetails and background information on the dataset and Kaggle competition ‘Plant Pathology 2020 Challenge’ were published as a peer-reviewed research article. If you use the dataset for your project, please cite the following\n\nhttps://bsapubs.onlinelibrary.wiley.com/doi/10.1002/aps3.11390","metadata":{}},{"cell_type":"markdown","source":"# Submission Format\nFor every author in the dataset, submission files should contain two columns: image and labels. labels should be a space-delimited list.\n\nThe file should contain a header and have the following format:\n\n* image, labels\n* 85f8cb619c66b863.jpg,healthy\n* ad8770db05586b59.jpg,healthy\n* c7b03e718489f3ca.jpg,healthy","metadata":{}},{"cell_type":"markdown","source":"### Prerequisites","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport PIL\nimport PIL.Image\nimport tensorflow as tf\nfrom keras.utils import to_categorical\nfrom keras.preprocessing import image\nfrom tqdm import tqdm\n!pip install tqdm\nimport csv\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dropout, Dense, Activation, Flatten, Conv2D, MaxPooling2D, BatchNormalization","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:35.535138Z","iopub.execute_input":"2021-05-26T10:38:35.535463Z","iopub.status.idle":"2021-05-26T10:38:47.891861Z","shell.execute_reply.started":"2021-05-26T10:38:35.535416Z","shell.execute_reply":"2021-05-26T10:38:47.890842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading data","metadata":{}},{"cell_type":"code","source":"train_dir= '../input/plant-pathology-2021-fgvc8/train_images'\ntest_dir =  '../input/plant-pathology-2021-fgvc8/test_images'\ntrain = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')\ntrain.head","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:47.893861Z","iopub.execute_input":"2021-05-26T10:38:47.894240Z","iopub.status.idle":"2021-05-26T10:38:47.947442Z","shell.execute_reply.started":"2021-05-26T10:38:47.894196Z","shell.execute_reply":"2021-05-26T10:38:47.946679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.DataFrame(train,columns = ['image','labels'])\ntrain['labels'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:47.950578Z","iopub.execute_input":"2021-05-26T10:38:47.950835Z","iopub.status.idle":"2021-05-26T10:38:47.965565Z","shell.execute_reply.started":"2021-05-26T10:38:47.950810Z","shell.execute_reply":"2021-05-26T10:38:47.964611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['labels'] = train['labels'].apply(lambda s: s.split(' '))\ntrain[:10]","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:47.967486Z","iopub.execute_input":"2021-05-26T10:38:47.967863Z","iopub.status.idle":"2021-05-26T10:38:47.998971Z","shell.execute_reply.started":"2021-05-26T10:38:47.967825Z","shell.execute_reply":"2021-05-26T10:38:47.998206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the Image Data Generator to import the images from the dataset\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(rescale = 1/255.,\n    rotation_range = 10,#Performing Rotation\n    zoom_range = 0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    validation_split= 0.2)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:48.000301Z","iopub.execute_input":"2021-05-26T10:38:48.000680Z","iopub.status.idle":"2021-05-26T10:38:48.005906Z","shell.execute_reply.started":"2021-05-26T10:38:48.000643Z","shell.execute_reply":"2021-05-26T10:38:48.004929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HEIGHT = 384\nWIDTH=384\nSEED = 98\nBATCH_SIZE=64\n","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:48.007453Z","iopub.execute_input":"2021-05-26T10:38:48.007809Z","iopub.status.idle":"2021-05-26T10:38:48.017370Z","shell.execute_reply.started":"2021-05-26T10:38:48.007772Z","shell.execute_reply":"2021-05-26T10:38:48.016609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/resized-plant2021/img_sz_384',# We are using the resized images otherwise it will take a lot of time to train \n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"training\",\n    color_mode=\"rgb\",\n    target_size = (HEIGHT,WIDTH),\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=SEED,\n)\n\n\nval_ds = datagen.flow_from_dataframe(\n    train,\n    directory = '../input/resized-plant2021/img_sz_384',# We are using the resized images otherwise it will take a lot of time to train \n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"validation\",\n    color_mode=\"rgb\",\n    target_size = (HEIGHT,WIDTH),\n    class_mode=\"categorical\",\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=SEED,\n)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:38:48.018805Z","iopub.execute_input":"2021-05-26T10:38:48.019613Z","iopub.status.idle":"2021-05-26T10:39:22.175241Z","shell.execute_reply.started":"2021-05-26T10:38:48.019573Z","shell.execute_reply":"2021-05-26T10:39:22.174323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = next(train_ds)\nprint(example[0].shape)\nplt.imshow(example[0][0,:,:,:])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:39:22.178030Z","iopub.execute_input":"2021-05-26T10:39:22.178405Z","iopub.status.idle":"2021-05-26T10:39:24.169949Z","shell.execute_reply.started":"2021-05-26T10:39:22.178372Z","shell.execute_reply":"2021-05-26T10:39:24.169156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=Sequential()\n\nmodel.add(Conv2D(32, (3, 3), padding=\"same\", activation='relu', input_shape=(HEIGHT, WIDTH,3)))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(MaxPooling2D(pool_size=(3, 3)))\nmodel.add(Dropout(0.25))\n        \nmodel.add(Conv2D(64, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(64, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=1))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(128, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(128, (3, 3), padding=\"same\", activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(64))\nmodel.add(Activation(\"relu\"))\nmodel.add(Dropout(0.25))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.25))\nmodel.add(Dense(6))\nmodel.add(Activation(\"softmax\"))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:39:24.171730Z","iopub.execute_input":"2021-05-26T10:39:24.172173Z","iopub.status.idle":"2021-05-26T10:39:26.525693Z","shell.execute_reply.started":"2021-05-26T10:39:24.172136Z","shell.execute_reply":"2021-05-26T10:39:26.524806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the Model\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.01, decay=0.01/30),\n    loss='binary_crossentropy',\n    metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:39:26.527143Z","iopub.execute_input":"2021-05-26T10:39:26.527560Z","iopub.status.idle":"2021-05-26T10:39:26.546503Z","shell.execute_reply.started":"2021-05-26T10:39:26.527518Z","shell.execute_reply":"2021-05-26T10:39:26.545637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint=ModelCheckpoint(r'Models\\model-x.h5',\n                          monitor='val_loss',\n                          mode='min',\n                          save_best_only=True,\n                          verbose=1)\ncallbacks=[checkpoint]","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:39:26.547665Z","iopub.execute_input":"2021-05-26T10:39:26.548071Z","iopub.status.idle":"2021-05-26T10:39:26.555921Z","shell.execute_reply.started":"2021-05-26T10:39:26.548027Z","shell.execute_reply":"2021-05-26T10:39:26.554563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_model=model.fit(train_ds,\n                    validation_data=val_ds,\n                    epochs=25,\n                    shuffle=True,\n                    verbose=1,\n                    batch_size=BATCH_SIZE,\n#                     steps_per_epoch=train_ds.samples//64,\n#                     validation_steps=val_ds.samples//64,\n                    callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T10:39:26.557469Z","iopub.execute_input":"2021-05-26T10:39:26.557922Z","iopub.status.idle":"2021-05-26T15:09:22.816884Z","shell.execute_reply.started":"2021-05-26T10:39:26.557836Z","shell.execute_reply":"2021-05-26T15:09:22.815978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cnn_model.save('model-cnn.h5')\nmodel.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:22.818641Z","iopub.execute_input":"2021-05-26T15:09:22.819054Z","iopub.status.idle":"2021-05-26T15:09:23.076998Z","shell.execute_reply.started":"2021-05-26T15:09:22.819006Z","shell.execute_reply":"2021-05-26T15:09:23.076145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:06:26.423056Z","iopub.execute_input":"2021-05-25T19:06:26.423433Z","iopub.status.idle":"2021-05-25T19:06:29.2149Z","shell.execute_reply.started":"2021-05-25T19:06:26.4234Z","shell.execute_reply":"2021-05-25T19:06:29.213697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = cnn_model.history\n\nplt.figure()\nplt.plot(model_history['accuracy'])\nplt.plot(model_history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'])\nplt.savefig('accuracy')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:23.078447Z","iopub.execute_input":"2021-05-26T15:09:23.078798Z","iopub.status.idle":"2021-05-26T15:09:23.277918Z","shell.execute_reply.started":"2021-05-26T15:09:23.078759Z","shell.execute_reply":"2021-05-26T15:09:23.277022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(model_history['loss'])\nplt.plot(model_history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'])\nplt.savefig('loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:23.279408Z","iopub.execute_input":"2021-05-26T15:09:23.279788Z","iopub.status.idle":"2021-05-26T15:09:23.457383Z","shell.execute_reply.started":"2021-05-26T15:09:23.279751Z","shell.execute_reply":"2021-05-26T15:09:23.456502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:23.458762Z","iopub.execute_input":"2021-05-26T15:09:23.459104Z","iopub.status.idle":"2021-05-26T15:09:23.479127Z","shell.execute_reply.started":"2021-05-26T15:09:23.459068Z","shell.execute_reply":"2021-05-26T15:09:23.478451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(\n    rescale = 1./255\n)\nINPUT_SIZE = (HEIGHT,WIDTH,3)\ntest_generator =  test_datagen.flow_from_dataframe(\n    submission,\n    directory=\"../input/plant-pathology-2021-fgvc8/test_images\",\n    x_col='image',\n    y_col=None,\n    class_mode=None,\n    target_size=INPUT_SIZE[:2]\n)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:23.480338Z","iopub.execute_input":"2021-05-26T15:09:23.480724Z","iopub.status.idle":"2021-05-26T15:09:23.492367Z","shell.execute_reply.started":"2021-05-26T15:09:23.480688Z","shell.execute_reply":"2021-05-26T15:09:23.491423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_generator)\nprint(preds)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:23.493888Z","iopub.execute_input":"2021-05-26T15:09:23.494247Z","iopub.status.idle":"2021-05-26T15:09:25.280096Z","shell.execute_reply.started":"2021-05-26T15:09:23.494209Z","shell.execute_reply":"2021-05-26T15:09:25.279246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = preds.tolist()\nindices = []\nfor pred in preds:\n    temp = []\n    for category in pred:\n        if category>=0.23:\n            temp.append(pred.index(category))\n    if temp!=[]:\n        indices.append(temp)\n    else:\n        temp.append(np.argmax(pred))\n        indices.append(temp)\n    \nprint(indices)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:25.281323Z","iopub.execute_input":"2021-05-26T15:09:25.281689Z","iopub.status.idle":"2021-05-26T15:09:25.291935Z","shell.execute_reply.started":"2021-05-26T15:09:25.281660Z","shell.execute_reply":"2021-05-26T15:09:25.291064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = (train_ds.class_indices)\nlabels = dict((v,k) for k,v in labels.items())\nprint(labels)\ntestlabels = []\nfor image in indices:\n    temp = []\n    for i in image:\n        temp.append(str(labels[i]))\n    testlabels.append(' '.join(temp))\nprint(testlabels)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:41.472668Z","iopub.execute_input":"2021-05-26T15:09:41.472997Z","iopub.status.idle":"2021-05-26T15:09:41.479032Z","shell.execute_reply.started":"2021-05-26T15:09:41.472966Z","shell.execute_reply":"2021-05-26T15:09:41.477880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['labels'] = testlabels\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:42.501107Z","iopub.execute_input":"2021-05-26T15:09:42.501506Z","iopub.status.idle":"2021-05-26T15:09:42.511800Z","shell.execute_reply.started":"2021-05-26T15:09:42.501467Z","shell.execute_reply":"2021-05-26T15:09:42.510560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T15:09:25.322365Z","iopub.execute_input":"2021-05-26T15:09:25.322641Z","iopub.status.idle":"2021-05-26T15:09:25.574846Z","shell.execute_reply.started":"2021-05-26T15:09:25.322617Z","shell.execute_reply":"2021-05-26T15:09:25.573973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}