{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nfrom glob import glob\nimport numpy as np\nfrom matplotlib import pyplot as plt\nimport math\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T07:01:47.513626Z","iopub.execute_input":"2022-08-10T07:01:47.514589Z","iopub.status.idle":"2022-08-10T07:01:47.715462Z","shell.execute_reply.started":"2022-08-10T07:01:47.514472Z","shell.execute_reply":"2022-08-10T07:01:47.714551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data loading","metadata":{}},{"cell_type":"code","source":"ScaleTo = 70  # px to scale\nseed = 7  # fixing random\n\npath = '../input/plant-seedlings-classification/train/*/*.png' \nfiles = glob(path)\n\ntrainImg = []\ntrainLabel = []\nj = 1\nnum = len(files)\n\n# Obtain images and resizing, obtain labels\nfor img in files:\n    print(str(j) + \"/\" + str(num), end=\"\\r\")\n    trainImg.append(cv2.resize(cv2.imread(img), (ScaleTo, ScaleTo)))  # Get image (with resizing)\n    trainLabel.append(img.split('/')[-2])  # Get image label (folder name)\n    j += 1\n\ntrainImg = np.asarray(trainImg)  # Train images set\ntrainLabel = pd.DataFrame(trainLabel)  # Train labels set","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:02:06.114089Z","iopub.execute_input":"2022-08-10T07:02:06.114720Z","iopub.status.idle":"2022-08-10T07:03:29.780386Z","shell.execute_reply.started":"2022-08-10T07:02:06.114680Z","shell.execute_reply":"2022-08-10T07:03:29.779424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show some example images\nfor i in range(8):\n    plt.subplot(2, 4, i + 1)\n    plt.imshow(trainImg[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:03:42.161115Z","iopub.execute_input":"2022-08-10T07:03:42.161466Z","iopub.status.idle":"2022-08-10T07:03:42.784634Z","shell.execute_reply.started":"2022-08-10T07:03:42.161436Z","shell.execute_reply":"2022-08-10T07:03:42.783647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Category count","metadata":{}},{"cell_type":"code","source":"trainLabel.groupby(by=0)[0].count()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:22:44.031087Z","iopub.execute_input":"2022-08-10T07:22:44.031678Z","iopub.status.idle":"2022-08-10T07:22:44.043941Z","shell.execute_reply.started":"2022-08-10T07:22:44.031642Z","shell.execute_reply":"2022-08-10T07:22:44.042928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Cleaning\n\n*     Use gaussian blur for remove noise\n*     Convert color to HSV\n*     Create mask\n*     Create boolean mask\n*     Apply boolean mask and getting image whithout background","metadata":{}},{"cell_type":"code","source":"clearTrainImg = []\nexamples = []; getEx = True\nfor img in trainImg:\n    # Use gaussian blur\n    blurImg = cv2.GaussianBlur(img, (5, 5), 0)   \n    \n    # Convert to HSV image\n    hsvImg = cv2.cvtColor(blurImg, cv2.COLOR_BGR2HSV)  \n    \n    # Create mask (parameters - green color range)\n    lower_green = (25, 40, 50)\n    upper_green = (75, 255, 255)\n    mask = cv2.inRange(hsvImg, lower_green, upper_green)  \n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (11, 11))\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)\n    \n    # Create bool mask\n    bMask = mask > 0  \n    \n    # Apply the mask\n    clear = np.zeros_like(img, np.uint8)  # Create empty image\n    clear[bMask] = img[bMask]  # Apply boolean mask to the origin image\n    \n    clearTrainImg.append(clear)  # Append image without backgroung\n    \n    # Show examples\n    if getEx:\n        plt.subplot(2, 3, 1); plt.imshow(img)  # Show the original image\n        plt.subplot(2, 3, 2); plt.imshow(blurImg)  # Blur image\n        plt.subplot(2, 3, 3); plt.imshow(hsvImg)  # HSV image\n        plt.subplot(2, 3, 4); plt.imshow(mask)  # Mask\n        plt.subplot(2, 3, 5); plt.imshow(bMask)  # Boolean mask\n        plt.subplot(2, 3, 6); plt.imshow(clear)  # Image without background\n        getEx = False\n\nclearTrainImg = np.asarray(clearTrainImg)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:23:05.766275Z","iopub.execute_input":"2022-08-10T07:23:05.766658Z","iopub.status.idle":"2022-08-10T07:23:08.193248Z","shell.execute_reply.started":"2022-08-10T07:23:05.766627Z","shell.execute_reply":"2022-08-10T07:23:08.192380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show sample result\nfor i in range(8):\n    plt.subplot(2, 4, i + 1)\n    plt.imshow(clearTrainImg[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:23:30.581841Z","iopub.execute_input":"2022-08-10T07:23:30.582436Z","iopub.status.idle":"2022-08-10T07:23:31.187548Z","shell.execute_reply.started":"2022-08-10T07:23:30.582401Z","shell.execute_reply":"2022-08-10T07:23:31.186482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train image Normalization","metadata":{}},{"cell_type":"code","source":"clearTrainImg = clearTrainImg / 255","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:23:56.083083Z","iopub.execute_input":"2022-08-10T07:23:56.083698Z","iopub.status.idle":"2022-08-10T07:23:56.245612Z","shell.execute_reply.started":"2022-08-10T07:23:56.083661Z","shell.execute_reply":"2022-08-10T07:23:56.244577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categories labeling and encoding","metadata":{}},{"cell_type":"code","source":"from keras.utils import np_utils\nfrom sklearn import preprocessing\nimport matplotlib.pyplot as plt\n\n# Encode labels and create classes\nle = preprocessing.LabelEncoder()\nle.fit(trainLabel[0])\nprint(\"Classes: \" + str(le.classes_))\nencodeTrainLabels = le.transform(trainLabel[0])\n\n# Make labels categorical\nclearTrainLabel = np_utils.to_categorical(encodeTrainLabels)\nnum_clases = clearTrainLabel.shape[1]\nprint(\"Number of classes: \" + str(num_clases))\n\n# Plot of label types numbers\ntrainLabel[0].value_counts().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:24:12.988560Z","iopub.execute_input":"2022-08-10T07:24:12.989130Z","iopub.status.idle":"2022-08-10T07:24:18.917588Z","shell.execute_reply.started":"2022-08-10T07:24:12.989095Z","shell.execute_reply":"2022-08-10T07:24:18.916479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset test and train split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrainX, testX, trainY, testY = train_test_split(clearTrainImg, clearTrainLabel, \n                                                test_size=0.1, random_state=seed, \n                                                stratify = clearTrainLabel)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:24:30.128271Z","iopub.execute_input":"2022-08-10T07:24:30.129249Z","iopub.status.idle":"2022-08-10T07:24:30.389457Z","shell.execute_reply.started":"2022-08-10T07:24:30.129212Z","shell.execute_reply":"2022-08-10T07:24:30.388433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\ndatagen = ImageDataGenerator(\n        rotation_range=180,  # randomly rotate images in the range\n        zoom_range = 0.1, # Randomly zoom image \n        width_shift_range=0.1,  # randomly shift images horizontally\n        height_shift_range=0.1,  # randomly shift images vertically \n        horizontal_flip=True,  # randomly flip images horizontally\n        vertical_flip=True  # randomly flip images vertically\n    )  \ndatagen.fit(trainX)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:24:48.992887Z","iopub.execute_input":"2022-08-10T07:24:48.993550Z","iopub.status.idle":"2022-08-10T07:24:49.190428Z","shell.execute_reply.started":"2022-08-10T07:24:48.993507Z","shell.execute_reply":"2022-08-10T07:24:49.189302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Builind Model (CNN)","metadata":{}},{"cell_type":"code","source":"import numpy\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Dropout\nfrom keras.layers import Flatten\nfrom keras.layers.convolutional import Conv2D\nfrom keras.layers.convolutional import MaxPooling2D\nfrom keras.layers import BatchNormalization\n\nfrom keras import backend as K\n\ndef recall_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    recall = true_positives / (possible_positives + K.epsilon())\n    return recall\n\ndef precision_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    return precision\n\ndef f1_m(y_true, y_pred):\n    precision = precision_m(y_true, y_pred)\n    recall = recall_m(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n\nnumpy.random.seed(seed)  # Fix seed\n\nmodel = Sequential()\n\nmodel.add(Conv2D(filters=64, kernel_size=(5, 5), input_shape=(ScaleTo, ScaleTo, 3), activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(filters=64, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Dropout(0.1))\n\nmodel.add(Conv2D(filters=128, kernel_size=(5, 5), activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(filters=128, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Dropout(0.1))\n\nmodel.add(Conv2D(filters=256, kernel_size=(5, 5), activation='relu'))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Conv2D(filters=256, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(BatchNormalization(axis=3))\nmodel.add(Dropout(0.1))\n\nmodel.add(Flatten())\n\nmodel.add(Dense(256, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\nmodel.add(Dense(256, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\n\nmodel.add(Dense(num_clases, activation='softmax'))\n\nmodel.summary()\n\n# compile model\nmodel.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['acc', f1_m, precision_m, recall_m])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:52:49.016694Z","iopub.execute_input":"2022-08-10T07:52:49.017669Z","iopub.status.idle":"2022-08-10T07:52:49.209873Z","shell.execute_reply.started":"2022-08-10T07:52:49.017620Z","shell.execute_reply":"2022-08-10T07:52:49.208770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Model","metadata":{}},{"cell_type":"code","source":"hist = model.fit(datagen.flow(trainX, trainY, batch_size=75), epochs=35, validation_data=(testX, testY))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:54:48.190024Z","iopub.execute_input":"2022-08-10T07:54:48.190390Z","iopub.status.idle":"2022-08-10T07:58:33.379385Z","shell.execute_reply.started":"2022-08-10T07:54:48.190359Z","shell.execute_reply":"2022-08-10T07:58:33.378415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model performance metrics","metadata":{}},{"cell_type":"code","source":"loss, accuracy, f1_score, precision, recall = model.evaluate(trainX, trainY)\nprint(\"\\nPerformance metrics for Train Data:\")\nprint(\"\\tLoss: {}\\n\\tAccuracy: {}\\n\\tF1: {}\\n\\tPrecision: {}\\n\\tRecall: {}\".format(loss, accuracy, f1_score, precision, recall))\n\nprint(\"\\n----------------------------------------\\n\")\n\nloss, accuracy, f1_score, precision, recall = model.evaluate(testX, testY)\nprint(\"\\nPerformance metrics for Test Data:\")\nprint(\"\\tLoss: {}\\n\\tAccuracy: {}\\n\\tF1: {}\\n\\tPrecision: {}\\n\\tRecall: {}\".format(loss, accuracy, f1_score, precision, recall))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:15:40.715395Z","iopub.execute_input":"2022-08-10T08:15:40.716431Z","iopub.status.idle":"2022-08-10T08:15:42.832383Z","shell.execute_reply.started":"2022-08-10T08:15:40.716391Z","shell.execute_reply":"2022-08-10T08:15:42.831414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Metrics","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport itertools\n\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \n    fig = plt.figure(figsize=(10,10))\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=90)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n\n# Predict the values from the validation dataset\npredY = model.predict(testX)\npredYClasses = np.argmax(predY, axis = 1) \ntrueY = np.argmax(testY, axis = 1) \n\n# confusion matrix\nconfusionMTX = confusion_matrix(trueY, predYClasses) \n\n# plot the confusion matrix\nplot_confusion_matrix(confusionMTX, classes = le.classes_) ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:00:42.267589Z","iopub.execute_input":"2022-08-10T08:00:42.268010Z","iopub.status.idle":"2022-08-10T08:00:43.341387Z","shell.execute_reply.started":"2022-08-10T08:00:42.267979Z","shell.execute_reply":"2022-08-10T08:00:43.340446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model prediction on Test data","metadata":{}},{"cell_type":"markdown","source":"### Loading data","metadata":{}},{"cell_type":"code","source":"path = '../input/plant-seedlings-classification/test/*.png'\nfiles = glob(path)\n\ntestImg = []\ntestId = []\nj = 1\nnum = len(files)\n\n# Obtain images and resizing, obtain labels\nfor img in files:\n    print(\"Obtain images: \" + str(j) + \"/\" + str(num), end='\\r')\n    testId.append(img.split('/')[-1])  # Images id's\n    testImg.append(cv2.resize(cv2.imread(img), (ScaleTo, ScaleTo)))\n    j += 1\n\ntestImg = np.asarray(testImg)  # Train images set\n\nfor i in range(8):\n    plt.subplot(2, 4, i + 1)\n    plt.imshow(testImg[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:17:51.649963Z","iopub.execute_input":"2022-08-10T08:17:51.650660Z","iopub.status.idle":"2022-08-10T08:17:55.708337Z","shell.execute_reply.started":"2022-08-10T08:17:51.650621Z","shell.execute_reply":"2022-08-10T08:17:55.707338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data cleaning","metadata":{}},{"cell_type":"code","source":"clearTestImg = []\nexamples = []; getEx = True\nfor img in testImg:\n    # Use gaussian blur\n    blurImg = cv2.GaussianBlur(img, (5, 5), 0)   \n    \n    # Convert to HSV image\n    hsvImg = cv2.cvtColor(blurImg, cv2.COLOR_BGR2HSV)  \n    \n    # Create mask (parameters - green color range)\n    lower_green = (25, 40, 50)\n    upper_green = (75, 255, 255)\n    mask = cv2.inRange(hsvImg, lower_green, upper_green)  \n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (11, 11))\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)\n    \n    # Create bool mask\n    bMask = mask > 0  \n    \n    # Apply the mask\n    clear = np.zeros_like(img, np.uint8)  # Create empty image\n    clear[bMask] = img[bMask]  # Apply boolean mask to the origin image\n    \n    clearTestImg.append(clear)  # Append image without backgroung\n    \n    # Show examples\n    if getEx:\n        plt.subplot(2, 3, 1); plt.imshow(img)  # Show the original image\n        plt.subplot(2, 3, 2); plt.imshow(blurImg)  # Blur image\n        plt.subplot(2, 3, 3); plt.imshow(hsvImg)  # HSV image\n        plt.subplot(2, 3, 4); plt.imshow(mask)  # Mask\n        plt.subplot(2, 3, 5); plt.imshow(bMask)  # Boolean mask\n        plt.subplot(2, 3, 6); plt.imshow(clear)  # Image without background\n        getEx = False\n\nclearTestImg = np.asarray(clearTestImg)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:17:56.732679Z","iopub.execute_input":"2022-08-10T08:17:56.733365Z","iopub.status.idle":"2022-08-10T08:17:57.676492Z","shell.execute_reply.started":"2022-08-10T08:17:56.733330Z","shell.execute_reply":"2022-08-10T08:17:57.675452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show sample result\nfor i in range(8):\n    plt.subplot(2, 4, i + 1)\n    plt.imshow(clearTestImg[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:17:58.433946Z","iopub.execute_input":"2022-08-10T08:17:58.434325Z","iopub.status.idle":"2022-08-10T08:17:59.087545Z","shell.execute_reply.started":"2022-08-10T08:17:58.434294Z","shell.execute_reply":"2022-08-10T08:17:59.086509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normalizing","metadata":{}},{"cell_type":"code","source":"clearTestImg = clearTestImg / 255","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:18:02.722173Z","iopub.execute_input":"2022-08-10T08:18:02.723294Z","iopub.status.idle":"2022-08-10T08:18:02.781004Z","shell.execute_reply.started":"2022-08-10T08:18:02.723227Z","shell.execute_reply":"2022-08-10T08:18:02.779183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Prediction","metadata":{}},{"cell_type":"code","source":"pred = model.predict(clearTestImg)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:18:07.530388Z","iopub.execute_input":"2022-08-10T08:18:07.531327Z","iopub.status.idle":"2022-08-10T08:18:07.811466Z","shell.execute_reply.started":"2022-08-10T08:18:07.531290Z","shell.execute_reply":"2022-08-10T08:18:07.810421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write result to file\npredNum = np.argmax(pred, axis=1)\npredStr = le.classes_[predNum]\n\nres = {'file': testId, 'species': predStr}\nres = pd.DataFrame(res)\nres.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:19:16.113241Z","iopub.execute_input":"2022-08-10T08:19:16.113696Z","iopub.status.idle":"2022-08-10T08:19:16.126750Z","shell.execute_reply.started":"2022-08-10T08:19:16.113662Z","shell.execute_reply":"2022-08-10T08:19:16.125527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}