{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"# importing required packages\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, GlobalAveragePooling2D, Conv2D, MaxPooling2D, Flatten\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras import metrics\nfrom tensorflow.keras.losses import BinaryCrossentropy\nfrom tensorflow.keras import applications","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> In this notebook lets classify the given data to healthy and unhealthy leaf"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv_path = '../input/cassava-leaf-disease-classification/train.csv'\nlabel_json_path = '../input/cassava-leaf-disease-classification/label_num_to_disease_map.json'\nimages_dir_path = '../input/cassava-leaf-disease-classification/train_images'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# READING THE DATA AND PROCESSING"},{"metadata":{"trusted":true},"cell_type":"code","source":"label_class_j = pd.read_json(label_json_path, orient='index')\nprint(\"The dataset have the following labels \")\nprint(label_class_j)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nchanging the labels to healthy - 1\n                       Unhealthy - 0\n'''\ntrain_csv = pd.read_csv(train_csv_path)\ntrain_csv['label'] = train_csv.label.map(lambda x: '1' if x==4 else '0')\nclss = [\"Unhealthy\", \"Healthy\"]\nclass_numbers = train_csv.label.value_counts().values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.bar(clss, train_csv.labels.value_counts())\nplt.title(\"Imbalanced data\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The class weights are calculated to penalize the loss in training due to imbalance in data\ntotal = len(train_csv)\nweight_of_0 = (total-class_numbers[0])/total\nweight_of_1 = (total-class_numbers[1])/total\nprint(f\"Weights of unhealthy : {weight_of_0}\\nWeight os healthy : {weight_of_1}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BATCH_SIZE = 32\nIMG_SIZE = 32\nEPOCHS = 16\nlr = 0.0005","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# LOADING THE DATA TO GENERATOR"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_gen = ImageDataGenerator(\n                                rotation_range=210,\n                                width_shift_range=0.1,\n                                height_shift_range=0.1,\n                                brightness_range=[0.1,0.9],\n                                shear_range=0.1,\n                                zoom_range=0.35,\n                                channel_shift_range=0.1,\n                                horizontal_flip=True,\n                                vertical_flip=True,\n                                rescale=1/255,\n                                validation_split=0.3\n                                \n                               )\n                                    \n    \nvalid_gen = ImageDataGenerator(rescale=1/255,\n                               validation_split = 0.3\n                              )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e4f272cf-ee78-47e3-b6da-f058dbdd3a08","_cell_guid":"34fa23a5-6377-45ef-9136-b3d47ac451e3","trusted":true},"cell_type":"code","source":"train_generator = train_gen.flow_from_dataframe(\n                            dataframe=train_csv,\n                            directory = images_dir_path,\n                            x_col = \"image_id\",\n                            y_col = \"label\",\n                            target_size = (IMG_SIZE, IMG_SIZE),\n                            class_mode = \"sparse\",\n                            batch_size = BATCH_SIZE,\n                            shuffle = True,\n                            subset = \"training\"\n\n)\n\nvalid_generator = valid_gen.flow_from_dataframe(\n                            dataframe=train_csv,\n                            directory = images_dir_path,\n                            x_col = \"image_id\",\n                            y_col = \"label\",\n                            target_size = (IMG_SIZE, IMG_SIZE),\n                            class_mode = \"sparse\",\n                            batch_size = BATCH_SIZE,\n                            shuffle = True,\n                            subset = \"validation\"\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# BUILDING & TRAINING MODEL"},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model():\n    model = tf.keras.Sequential([\n        Conv2D(32, (3,3), padding='SAME', activation='relu', input_shape=[IMG_SIZE,IMG_SIZE,3]),\n        Conv2D(64, (3,3), padding='SAME', activation='relu'),\n        Conv2D(64, (3,3), padding='SAME', activation='relu'),\n        MaxPooling2D(),\n        Conv2D(64, (3,3), padding='SAME', activation='relu'),\n        Conv2D(64, (3,3), padding='SAME', activation='relu'),\n        MaxPooling2D(),\n        Conv2D(32, (3,3), padding='SAME', activation='relu'),\n        Conv2D(32, (3,3), padding='SAME', activation='relu'),\n        Conv2D(32, (3,3), padding='SAME', activation='relu'),\n        Conv2D(32, (3,3), padding='VALID', activation='relu'),\n        Dense(256, activation='relu'),\n        Flatten(),\n        Dense(1, activation='sigmoid')\n        \n    ])\n    \n    model.compile(loss=BinaryCrossentropy(), optimizer=Adam(learning_rate=lr), metrics=['acc'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = build_model()\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Callback for saving the best model during training\ncallback0 = tf.keras.callbacks.ModelCheckpoint(\"./CasavaLeafDisease_BiModel.h5\", monitor='val_loss',save_best_only=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# class_weight is added to penalize tha loss during training as the data set is imbalanced\nhis = model.fit(train_generator, validation_data=valid_generator, epochs=EPOCHS, callbacks=[callback0], class_weight={0:weight_of_0, 1:weight_of_1})","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"I have used class_weight parameter in fit() as the data is imbalanced but, why the validation accuracy is same all time ?\nAnyone help me with this. Am I doing it right or not?"},{"metadata":{},"cell_type":"markdown","source":"The further work is continued on below notebook. check it out.\nhttps://www.kaggle.com/manojkumars00/casava-leaf-disease-simple-classification"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}