{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Apple Disease Classification","metadata":{}},{"cell_type":"code","source":"!pip install natsort","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:08.273617Z","iopub.execute_input":"2022-07-22T11:23:08.274660Z","iopub.status.idle":"2022-07-22T11:23:22.346083Z","shell.execute_reply.started":"2022-07-22T11:23:08.274543Z","shell.execute_reply":"2022-07-22T11:23:22.344702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import Required Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport random\nimport glob\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom re import search\nimport shutil\nimport natsort\nfrom PIL import Image\nfrom tqdm import tqdm\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Activation, Dropout,Flatten, Conv2D, MaxPooling2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping\nfrom tensorflow.keras.preprocessing import image\n\nimport warnings\nwarnings.filterwarnings('ignore')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:22.349315Z","iopub.execute_input":"2022-07-22T11:23:22.349974Z","iopub.status.idle":"2022-07-22T11:23:27.237819Z","shell.execute_reply.started":"2022-07-22T11:23:22.349926Z","shell.execute_reply":"2022-07-22T11:23:27.236821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define Path \nMAIN_PATH = os.getcwd()\nIMAGE_DIR = '../input/plant-pathology-2020-fgvc7/images'\n# Image Size for Data Generator\nIMG_SIZE = 224\n# Batch Size\nBATCH_SIZE = 16\n# Epochs\nEPOCHS = 50","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:27.239145Z","iopub.execute_input":"2022-07-22T11:23:27.239838Z","iopub.status.idle":"2022-07-22T11:23:27.247380Z","shell.execute_reply.started":"2022-07-22T11:23:27.239798Z","shell.execute_reply":"2022-07-22T11:23:27.246480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Data","metadata":{}},{"cell_type":"code","source":"# Train dataset\ntrain_df = pd.read_csv(\"../input/plant-pathology-2020-fgvc7/train.csv\")\n# Test dataset\ntest_df = pd.read_csv(\"../input/plant-pathology-2020-fgvc7/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:27.249776Z","iopub.execute_input":"2022-07-22T11:23:27.250527Z","iopub.status.idle":"2022-07-22T11:23:27.285160Z","shell.execute_reply.started":"2022-07-22T11:23:27.250479Z","shell.execute_reply":"2022-07-22T11:23:27.284253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check the glimpse of first five rows of train dataset\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:27.286726Z","iopub.execute_input":"2022-07-22T11:23:27.287049Z","iopub.status.idle":"2022-07-22T11:23:27.305635Z","shell.execute_reply.started":"2022-07-22T11:23:27.287016Z","shell.execute_reply":"2022-07-22T11:23:27.304798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check some images \nrandom_images = random.sample(glob.glob(os.path.join(IMAGE_DIR,\"Train_*.jpg\")),10)\nrandom_id = [img_path.split('/')[-1] for img_path in random_images]\n\ndef convert_dummies_to_category(df):\n    data = df.copy()\n    s2 = data[['healthy','multiple_diseases','rust','scab']].idxmax(axis=1)\n    data = pd.concat([data,s2],axis=1)\n    data.rename({0:'category'},axis=1,inplace=True)\n    return data\n  \ndf_copy = convert_dummies_to_category(train_df)\n\ndef get_category_img(img):\n    ''' function to fetch the category of the image'''\n    if search(\"Train\",img):\n        img = img.split('.')[0]\n        category = df_copy.loc[df_copy['image_id']==img]['category']\n        return category\n\nplt.figure(figsize=(18, 6))\nfor idx, img_path in enumerate(random_id):\n    img_title = get_category_img(img_path)\n    sp = plt.subplot(2, 5, idx+1)\n    sp.axis('Off')\n    complete_path = os.path.join(IMAGE_DIR,img_path)\n    mp_image = mpimg.imread(complete_path)\n    plt.title(img_title.values[0])\n    plt.imshow(mp_image) ","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:27.308336Z","iopub.execute_input":"2022-07-22T11:23:27.308617Z","iopub.status.idle":"2022-07-22T11:23:31.502309Z","shell.execute_reply.started":"2022-07-22T11:23:27.308592Z","shell.execute_reply":"2022-07-22T11:23:31.501329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Preparation","metadata":{}},{"cell_type":"code","source":"# data labels - Target  \nlabels = train_df.loc[:,'healthy':].columns\nprint(labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.503297Z","iopub.execute_input":"2022-07-22T11:23:31.503621Z","iopub.status.idle":"2022-07-22T11:23:31.511118Z","shell.execute_reply.started":"2022-07-22T11:23:31.503590Z","shell.execute_reply":"2022-07-22T11:23:31.510053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets give labels as following to separate the images into different folder as per their labels\n# 'health':0 , 'multiple_diseases':1, 'rust':2, 'scab':3\n\ndef label_encoder(df, labels):\n    label_counter = 0\n    df['label'] = 0\n    for class_name in labels:\n        df['label'] = df['label'] + df[class_name] * label_counter\n        label_counter += 1\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.513093Z","iopub.execute_input":"2022-07-22T11:23:31.514025Z","iopub.status.idle":"2022-07-22T11:23:31.519923Z","shell.execute_reply.started":"2022-07-22T11:23:31.513990Z","shell.execute_reply":"2022-07-22T11:23:31.518834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply above function to create a label for each class\ntrain_df = label_encoder(train_df, labels)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.521811Z","iopub.execute_input":"2022-07-22T11:23:31.522592Z","iopub.status.idle":"2022-07-22T11:23:31.540159Z","shell.execute_reply.started":"2022-07-22T11:23:31.522556Z","shell.execute_reply":"2022-07-22T11:23:31.539235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_label_img(img):\n    ''' function to fetch the label of the image'''\n    if search(\"Train\",img):\n        img=img.split('.')[0]\n        label = train_df.loc[train_df['image_id']==img]['label']\n        return label","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.544214Z","iopub.execute_input":"2022-07-22T11:23:31.544870Z","iopub.status.idle":"2022-07-22T11:23:31.550372Z","shell.execute_reply.started":"2022-07-22T11:23:31.544834Z","shell.execute_reply":"2022-07-22T11:23:31.549435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets separate the images into separate folders according to their labels\ndef create_dir_image_labels():\n    # sort the filenames in IMAGE_DIR\n    images = natsort.natsorted(os.listdir(IMAGE_DIR))\n    for img in tqdm(images):\n        # function to fetch the label of the image\n        label = get_label_img(img)\n        # image path from the original image dir\n        image_path = os.path.join(IMAGE_DIR,img)\n        \n        if search(\"Train\",img):\n            # if image label is equal to 0, copy that image into healthy directory.\n            if (img.split(\"_\")[1].split(\".\")[0]) and label.item()==0:\n                shutil.copy(image_path, os.path.join(MAIN_PATH,'train','healthy'))\n             \n            # if image label is equal to 1, copy that image into multiple_diseases directory.\n            elif(img.split(\"_\")[1].split(\".\")[0]) and label.item()==1:\n                shutil.copy(image_path,os.path.join(MAIN_PATH,'train','multiple_diseases'))\n               \n            # if image label is equal to 2, copy that image into rust directory.\n            elif(img.split(\"_\")[1].split(\".\")[0]) and label.item()==2:\n                shutil.copy(image_path,os.path.join(MAIN_PATH,'train','rust'))\n             \n            # if image label is equal to 3, copy that image into scab directory.\n            elif(img.split(\"_\")[1].split(\".\")[0]) and label.item()==3:\n                shutil.copy(image_path,os.path.join(MAIN_PATH,'train','scab'))\n         \n        # else copy all excluded images from the above conditions into test directory.\n        elif search(\"Test\",img):\n            shutil.copy(image_path,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.552438Z","iopub.execute_input":"2022-07-22T11:23:31.553683Z","iopub.status.idle":"2022-07-22T11:23:31.565287Z","shell.execute_reply.started":"2022-07-22T11:23:31.553647Z","shell.execute_reply":"2022-07-22T11:23:31.564295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create directory for train - healthy, multiple_diseases, rust, scab and test images\nshutil.os.mkdir(os.path.join(MAIN_PATH, 'train'))\nshutil.os.mkdir(os.path.join(MAIN_PATH,'train','healthy'))\nshutil.os.mkdir(os.path.join(MAIN_PATH,'train','multiple_diseases'))\nshutil.os.mkdir(os.path.join(MAIN_PATH,'train','rust'))\nshutil.os.mkdir(os.path.join(MAIN_PATH,'train','scab'))\n\nshutil.os.mkdir(os.path.join(MAIN_PATH, 'test'))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.566857Z","iopub.execute_input":"2022-07-22T11:23:31.567752Z","iopub.status.idle":"2022-07-22T11:23:31.578312Z","shell.execute_reply.started":"2022-07-22T11:23:31.567711Z","shell.execute_reply":"2022-07-22T11:23:31.577272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply function to copy images belongs to their respective labels\ncreate_dir_image_labels()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:23:31.581284Z","iopub.execute_input":"2022-07-22T11:23:31.581977Z","iopub.status.idle":"2022-07-22T11:24:11.348516Z","shell.execute_reply.started":"2022-07-22T11:23:31.581940Z","shell.execute_reply":"2022-07-22T11:24:11.347547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Augmentation","metadata":{}},{"cell_type":"code","source":"# All images will be rescaled by 1.0/255\ndatagen=ImageDataGenerator(rescale=1.0/255,\n                                shear_range=0.2,\n                                zoom_range=0.2,\n                                horizontal_flip=True,\n                                vertical_flip=True,\n                                validation_split=0.2)\n\n# Path - Train Directory \nTrain_DIR = os.path.join(MAIN_PATH, 'train')\n\n# Flow training images in batches of 16 using datagen generator\ntrain_datagen=datagen.flow_from_directory(Train_DIR,\n                                         target_size=(IMG_SIZE,IMG_SIZE),\n                                         batch_size=BATCH_SIZE,\n                                         class_mode='categorical',\n                                         subset='training')\n\n# Flow validation images in batches of 16 using datagen generator\nval_datagen=datagen.flow_from_directory(Train_DIR,\n                                         target_size=(IMG_SIZE,IMG_SIZE),\n                                         batch_size=BATCH_SIZE,\n                                         class_mode='categorical',\n                                         subset='validation')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:24:11.349845Z","iopub.execute_input":"2022-07-22T11:24:11.350763Z","iopub.status.idle":"2022-07-22T11:24:11.565701Z","shell.execute_reply.started":"2022-07-22T11:24:11.350723Z","shell.execute_reply":"2022-07-22T11:24:11.564656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Building","metadata":{}},{"cell_type":"code","source":"# Formation of CNN model\n\nmodel=Sequential()\n# Note the input shape is the desired size of the image 224x224 with 3 color\nmodel.add(Conv2D(64,(3,3),activation='relu',padding='same',input_shape=(IMG_SIZE,IMG_SIZE,3)))\n# Max Pooling Layer\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Conv2D(64,(3,3),activation='relu',padding='same'))\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Conv2D(64,(3,3),activation='relu',padding='same'))\nmodel.add(MaxPooling2D(2,2))\nmodel.add(Conv2D(128,(3,3),activation='relu',padding='same'))\nmodel.add(MaxPooling2D(2,2))\n# Flatten the results to feed into a DNN\nmodel.add(Flatten())\n# 4 output neuron because of 4 labels in output layer\nmodel.add(Dense(4,activation='softmax'))\n\n# Compile the Model\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='categorical_crossentropy',\n    metrics=['accuracy'])\n# Model Summary\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:24:11.567310Z","iopub.execute_input":"2022-07-22T11:24:11.567671Z","iopub.status.idle":"2022-07-22T11:24:14.164261Z","shell.execute_reply.started":"2022-07-22T11:24:11.567634Z","shell.execute_reply":"2022-07-22T11:24:14.163309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Defining Callbacks","metadata":{}},{"cell_type":"code","source":"# Checkpoint to save the best model on the basis of validation loss\ncheckpoint=ModelCheckpoint('./best_weights.h5',\n                          monitor='val_loss',\n                          mode='min',\n                          save_best_only=True,\n                          verbose=1)\n# Early Stop - if the number of epochs that is 10 without improvement of validation loss after which training will be early stopped.\nearlystop=EarlyStopping(monitor='val_loss',\n                       min_delta=0,\n                       patience=10,\n                       verbose=1,\n                       restore_best_weights=True)\n\ncallbacks=[checkpoint,earlystop]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:24:14.166304Z","iopub.execute_input":"2022-07-22T11:24:14.166977Z","iopub.status.idle":"2022-07-22T11:24:14.174158Z","shell.execute_reply.started":"2022-07-22T11:24:14.166938Z","shell.execute_reply":"2022-07-22T11:24:14.173171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(train_datagen,validation_data=val_datagen,\n                                 epochs=EPOCHS,\n                                 steps_per_epoch=train_datagen.samples//BATCH_SIZE,\n                                 validation_steps=val_datagen.samples//BATCH_SIZE,\n                                 callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T11:24:14.176165Z","iopub.execute_input":"2022-07-22T11:24:14.176593Z","iopub.status.idle":"2022-07-22T12:17:11.500356Z","shell.execute_reply.started":"2022-07-22T11:24:14.176556Z","shell.execute_reply":"2022-07-22T12:17:11.498922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Result","metadata":{}},{"cell_type":"code","source":"plt.figure(1, figsize = (20, 6))\n\nplt.subplot(121)\nplt.plot(history.history['loss'], label = 'Train')\nplt.plot(history.history['val_loss'], label = 'Validation')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.subplot(122)\nplt.plot(history.history['accuracy'], label = 'Train')\nplt.plot(history.history['val_accuracy'], label = 'Validation')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:17:11.502036Z","iopub.execute_input":"2022-07-22T12:17:11.502390Z","iopub.status.idle":"2022-07-22T12:17:11.834118Z","shell.execute_reply.started":"2022-07-22T12:17:11.502353Z","shell.execute_reply":"2022-07-22T12:17:11.833166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction","metadata":{}},{"cell_type":"code","source":"def check_apple_leaf(image_name):\n    DIR = os.path.join(IMAGE_DIR, image_name)\n    image_result = Image.open(DIR)\n    \n    test_image = image.load_img(DIR,target_size=(IMG_SIZE,IMG_SIZE))\n    test_image = image.img_to_array(test_image)\n    test_image = test_image/255\n    test_image = np.expand_dims(test_image,axis=0)\n    result = model.predict(test_image)\n\n    Categories = ['healthy','multiple_disease','rust','scab']\n    image_result = plt.imshow(image_result)\n    plt.title(Categories[np.argmax(result)])\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:17:11.835634Z","iopub.execute_input":"2022-07-22T12:17:11.836234Z","iopub.status.idle":"2022-07-22T12:17:11.844264Z","shell.execute_reply.started":"2022-07-22T12:17:11.836181Z","shell.execute_reply":"2022-07-22T12:17:11.843136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_apple_leaf('Test_1002.jpg')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:17:11.845753Z","iopub.execute_input":"2022-07-22T12:17:11.846268Z","iopub.status.idle":"2022-07-22T12:17:12.608340Z","shell.execute_reply.started":"2022-07-22T12:17:11.846230Z","shell.execute_reply":"2022-07-22T12:17:12.607389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_apple_leaf('Test_1010.jpg')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:17:58.368081Z","iopub.execute_input":"2022-07-22T12:17:58.368438Z","iopub.status.idle":"2022-07-22T12:17:58.960592Z","shell.execute_reply.started":"2022-07-22T12:17:58.368407Z","shell.execute_reply":"2022-07-22T12:17:58.959729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_apple_leaf('Test_1018.jpg')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T12:19:04.518921Z","iopub.execute_input":"2022-07-22T12:19:04.519274Z","iopub.status.idle":"2022-07-22T12:19:05.104767Z","shell.execute_reply.started":"2022-07-22T12:19:04.519245Z","shell.execute_reply":"2022-07-22T12:19:05.103855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}