{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport tensorflow as tf\nimport keras\nimport os\n\n# for filename in filenames:\n#         print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sb_weed =   '../input/weed-detection-in-soybean-crops/dataset/'\nws =        '../input/v2-plant-seedlings-dataset/'\ncas_d =     '../input/cassava-leaf-disease-classification/train_images/'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Defining empty lists to store filenames and their equivalent class names for weed seedlings\nfns = []\nclss = []\n\n\n#Defining empty lists to store filenames and their equivalent class names for soyabean/weeds\nfns2 = []\nclss2 = []\n\n#Defining empty lists to store filenames and their equivalent class names for cassava\nfns3 = []\nclss3 = []","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Batch Meta data extraction from images\n\nab = []\nac = []\nfor i in os.listdir(ws +\"Black-grass\"):\n   \n        ab.append(ws +\"Black-grass\"+\"/\"+i)\n        ac.append(\"Black-grass\")\n        \n\n\nfor i in os.listdir(ws +\"Charlock\"):\n   \n        ab.append(ws +\"Charlock\"+\"/\"+i)\n        ac.append(\"Charlock\")\n        \n\n\n\nfor i in os.listdir(ws +\"Cleavers\"):\n   \n        ab.append(ws +\"Cleavers\"+\"/\"+i)\n        ac.append(\"Cleavers\")\n        \n\n\nfor i in os.listdir(ws +\"Common Chickweed\"):\n   \n        ab.append(ws +\"Common Chickweed\"+\"/\"+i)\n        ac.append(\"Common Chickweed\")\n        \n\n\n\nfor i in os.listdir(ws +\"Common wheat\"):\n   \n        ab.append(ws +\"Common wheat\"+\"/\"+i)\n        ac.append(\"Common wheat\")\n        \n\nfor i in os.listdir(ws +\"Fat Hen\"):\n   \n        ab.append(ws +\"Fat Hen\"+\"/\"+i)\n        ac.append(\"Fat Hen\")\n        \n\nfor i in os.listdir(ws +\"Loose Silky-bent\"):\n   \n        ab.append(ws +\"Loose Silky-bent\"+\"/\"+i)\n        ac.append(\"Loose Silky-bent\")\n\n\nfor i in os.listdir(ws +\"Maize\"):\n   \n        ab.append(ws +\"Maize\"+\"/\"+i)\n        ac.append(\"Maize\")\n\n        \nfor i in os.listdir(ws +\"Scentless Mayweed\"):\n   \n        ab.append(ws +\"Scentless Mayweed\"+\"/\"+i)\n        ac.append(\"Scentless Mayweed\")\n               \n            \n            \nfor i in os.listdir(ws +\"Shepherd’s Purse\"):\n   \n        ab.append(ws +\"Shepherd’s Purse\"+\"/\"+i)\n        ac.append(\"Shepherd’s Purse\")\n        \n        \n\nfor i in os.listdir(ws +\"Small-flowered Cranesbill\"):\n   \n        ab.append(ws +\"Small-flowered Cranesbill\"+\"/\"+i)\n        ac.append(\"Small-flowered Cranesbill\")\n        \n        \n\nfor i in os.listdir(ws +\"Sugar beet\"):\n   \n        ab.append(ws +\"Sugar beet\" + \"/\" +i)\n        ac.append(\"Sugar beet\")\n                \n            \n\n# for i in os.listdir(ws +\"nonsegmentedv2\"):\n   \n#         ab.append(ws +\"nonsegmentedv2\" + \"/\" +i)\n#         ac.append(\"nonsegmentedv2\")\n        \n    \nlen(ab), len(ac)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # For soybean meta data extraction\n# for i in os.listdir(sb_weed +\"broadleaf\"):\n   \n#         ab.append(ws +\"broadleaf\" + \"/\" +i)\n#         ac.append(\"broadleaf\")\n        \n        \n        \n# for i in os.listdir(sb_weed +\"grass\"):\n   \n#         ab.append(ws +\"grass\" + \"/\" +i)\n#         ac.append(\"grass\")\n        \n\n# for i in os.listdir(sb_weed +\"soil\"):\n   \n#         ab.append(ws +\"soil\" + \"/\" +i)\n#         ac.append(\"soil\")\n        \n        \n# for i in os.listdir(sb_weed +\"soybean\"):\n   \n#         ab.append(ws +\"soybean\" + \"/\" +i)\n#         ac.append(\"soybean\")\n        \n# len(ab), len(ac)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading in cassava metadata\n\ncas_d = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\n\ncas_d.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cas_d['label']= cas_d['label'].map({0:\"Cassava Bacterial Blight (CBB)\", \n                                          1:\"Cassava Brown Streak Disease (CBSD)\",\n                                          2:\"Cassava Green Mottle (CGM)\",\n                                          3:\"Cassava Mosaic Disease (CMD)\",\n                                          4:\"Healthy\"\n                                        \n                                          \n                                          }, na_action= 'ignore')\n\ncas_d.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extracting  cassava metadata to list\n\nfor i in cas_d['image_id']:\n    \n    fns3.append('../input/cassava-leaf-disease-classification/train_images/'+i)\n    \n    \nfor t in cas_d['label']:\n    clss3.append(t)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Below are the list of 13 classes contained in the Weed seedling dataset:\n\n1. Scentless Mayweed\n2. Common wheat\n3. nonsegmentedv2\n4. Charlock\n5. Black-grass\n6. Sugar beet\n7. Loose Silky-bent\n8. Maize\n9. Cleavers\n10. Common Chickweed\n11. Fat Hen\n12. Small-flowered Cranesbill\n13. Shepherd’s Purse\n","metadata":{}},{"cell_type":"code","source":"#create new dataframe for the extracted metadatas from the 3 different datasets\ndf1 = pd.DataFrame({'file_id':ab, \"label\":ac})\ndf3 = pd.DataFrame({'file_id':fns3, \"label\":clss3 })\n\n#merging all dataframes together\ndf = pd.concat([df1, df3], axis = 0)\n\n\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shapes comparison and balances\ndf.shape, len(fns3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking our number of unique labels and classes\ndf['label'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df['label'].unique()) #23 classes in total.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import roc_curve, roc_auc_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#splitting dataset for training and evaluation\n\nRANDOM_SEED = 7\n\nX_train, X_eval, y_train, y_eval = train_test_split(\n    df,\n    df['label'],\n    test_size=0.25,\n    shuffle=True,\n    stratify=df['label'],\n    random_state=RANDOM_SEED\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extracting digital footprints from the images\n\nfrom tensorflow.keras.preprocessing import image\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nimg_height = 100\nimg_width = 100\n\nX = []\n\nfor i in tqdm(df['file_id']):\n    img = image.load_img(i, target_size=(img_height, img_width, 3))\n    img = image.img_to_array(img)\n    img = img/255.0\n    X.append(img)\n    \nX = np.array(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Encoding the target variable for a multiclass-classification\nfrom sklearn.preprocessing import LabelEncoder\n# creating labelencoder instance\nlabelencoder = LabelEncoder()# Assigning numerical values and storing in another column\ny = labelencoder.fit_transform(df['label'])\n\ny","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building the Models","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense, Dropout, BatchNormalization, Conv2D, MaxPool2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing import image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## APPLYING THE RESNET-32 ARCHITECTURE","metadata":{}},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),padding='same', activation='relu', input_shape= X[0].shape))\nmodel.add(Conv2D(filters=64,kernel_size=(3,3),padding='same', activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(pool_size=(2,2), strides=2, padding='valid'))\nmodel.add(Dropout(0.5))\n\nmodel.add(Conv2D(filters=32,kernel_size=(3,3),padding='same', activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(pool_size=(2,2), strides=2, padding='valid'))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=32,kernel_size=(3,3),padding='same', activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(pool_size=(2,2), strides=2, padding='valid'))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=32,kernel_size=(3,3),padding='same', activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(pool_size=(2,2), strides=2, padding='valid'))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=32,kernel_size=(3,3),padding='same', activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(pool_size=(2,2), strides=2, padding='valid'))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=32,kernel_size=(3,3),padding='same', activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool2D(pool_size=(2,2), strides=2, padding='valid'))\nmodel.add(Dropout(0.2))\n\nmodel.add(Flatten())\nmodel.add(Dense(units=128, activation='relu'))\nmodel.add(Dense(units=17, activation='softmax'))\n\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='Adam',loss=keras.losses.SparseCategoricalCrossentropy(), metrics=[tf.keras.metrics.SparseCategoricalAccuracy()])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x=X, y=y, epochs=10, batch_size =500, verbose=1, validation_split = 0.15)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting Training and validation Accuracy values\nepoch_range = range(1,11)\nplt.plot(epoch_range, history.history['sparse_categorical_accuracy'])\nplt.plot(epoch_range, history.history['val_sparse_categorical_accuracy'])\nplt.title('Model_accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train','val'], loc='upper left')\nplt.show()\n\n#plot training and validation loss values\nplt.plot(epoch_range, history.history['loss'])\nplt.plot(epoch_range, history.history['val_loss'])\nplt.title('Model loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train','val'], loc='upper left')\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}