{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Importing Libraries"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom shutil import rmtree, copy\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom PIL import Image\nfrom matplotlib import pyplot as plt\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import silhouette_score\nfrom sklearn.cluster import KMeans, SpectralClustering","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Setting up the environment"},{"metadata":{"trusted":true},"cell_type":"code","source":"# OS variables\nseed = np.random.randint(0, 115)\n\n# Data path\npath = '../input/cassava-leaf-disease-classification/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(path + 'train.csv')\ndata['img_path'] = path + 'train_images/' + data.image_id\n\nlabel_dict_ = {\n    '0': 'Cassava Bacterial Blight (CBB)',\n    '1': 'Cassava Brown Streak Disease (CBSD)',\n    '2': 'Cassava Green Mottle (CGM)',\n    '3': 'Cassava Mosaic Disease (CMD)',\n    '4': 'Healthy'\n}\n\ndata['class_label'] = [label_dict_[str(x)] for x in data['label']]\n\nle = LabelEncoder()\ndata['class_'] = le.fit_transform(data['class_label'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Defining Parameters"},{"metadata":{"trusted":true},"cell_type":"code","source":"IMAGE_SIZE = 224\nBATCH_SIZE = 8\nMODEL_IMAGE_SIZE = 512\nEPOCHS = 20","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Removing Duplicates and Mis-labelled Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"data = data[~data['image_id'].isin(['1562043567.jpg', '3551135685.jpg', '2252529694.jpg'])]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Clustering Analysis"},{"metadata":{},"cell_type":"markdown","source":"### 1. Feature Extraction Baseline"},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_base_model = tf.keras.applications.ResNet50(include_top = False, weights = 'imagenet', input_shape = (IMAGE_SIZE, IMAGE_SIZE, 3))\nfeature_model_pooling = tf.keras.layers.GlobalAveragePooling2D()(feature_base_model.output)\n\nfeature_model = tf.keras.models.Model(inputs = feature_base_model.input, outputs = feature_model_pooling)\n#feature_model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 2. Reading Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"im_paths = data['img_path']\nclas = data['class_']\n#x_images = np.array([np.float32(Image.open(im_path).resize((IMAGE_SIZE, IMAGE_SIZE))) / 255.0 for im_path in im_paths])\n#Y = np.array([class_ for class_ in clas]) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(data['img_path'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 3. Predicting"},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = []\nfor i, im_path in enumerate(im_paths):\n    #print(\"Doing \" + str(i+1) )\n    x_images = np.array([np.float32(Image.open(im_path).resize((IMAGE_SIZE, IMAGE_SIZE))) / 255.0])\n    a = feature_model.predict(x_images)\n    #print(a[0])\n    #print(a[0].shape)\n    y_pred.append(a[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = np.array(y_pred)\nprint('y: {}'.format(y_pred.shape))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 4. Clustering using KMeans"},{"metadata":{"trusted":true},"cell_type":"code","source":"\"\"\"max_clusters = 5\n\nplt.figure(figsize=(10, 5))\nplt.style.use('ggplot')\n\nskip = 1\nfor K in range(2, max_clusters+1):\n    KMC = KMeans(n_clusters=K).fit(y_pred)\n    labels = KMC.labels_\n    print('fitting for {} clusters completed..'.format(K))\n    score = silhouette_score(y_pred, labels, metric='euclidean')\n    print('silhouette_score for {} clusters: {}'.format(K, score))\n    plt.plot(K, score, '^')\n\nplt.xlabel('K')\nplt.ylabel('Silhoutte Score')\nplt.show()\"\"\"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 5. Saving the Clusters"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Select the value of K based on silhouette_score\nK = 2\n\nKMC = KMeans(n_clusters=K, n_jobs=-1, random_state=seed)\nKMC.fit(y_pred)\nK_pred = KMC.predict(y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pickle\n\n# Save to file in the current working directory\npkl_filename = \"./Kmeans.pkl\"\nwith open(pkl_filename, 'wb') as file:\n    pickle.dump(KMC, file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"try:\n    rmtree('Clusters/')\n    os.mkdir('Clusters/')\nexcept: pass\n\nfor i in range(K):\n    os.makedirs('Clusters/' + str(i))\n    [os.makedirs('Clusters/{}/{}'.format(i, class_)) for class_ in list(label_dict_.values())]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i, im_path in enumerate(im_paths):\n    class_ = data[data['img_path'] == im_path].class_label.values[0]\n    copy(im_path, 'Clusters/{}/{}'.format(K_pred[i], class_))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!zip -r Clusters_ResNet.zip ./Clusters/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_images = None\nY = None\ny_pred = None","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Training Model for Cluster 0"},{"metadata":{},"cell_type":"markdown","source":"### 1. Augumentation and Preprocessing"},{"metadata":{"trusted":true},"cell_type":"code","source":"def preprocess(image):\n    #Converting to numpy array from numpy tensor with rank 3\n    image = np.array(image, dtype=np.uint8)\n    #Converting to RGB\n    #img = cv2.cvtCoor(img, cv2.COLOR_BGR2RGB)\n    #Gaussian Blur\n    gaussian_blur = cv2.GaussianBlur(image,(3,3),0)\n    img = np.asarray(gaussian_blur, dtype=np.float64)\n    return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dir = './Clusters/0'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del feature_model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BATCH_SIZE = 4","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Training  Augumentation\ndatagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1.0/255,\n                             rotation_range=30,\n                             zoom_range=0.3,\n                             horizontal_flip=True,\n                             brightness_range=[0.6, 1.2],\n                             validation_split=0.2,\n                             fill_mode='nearest',\n                             preprocessing_function=preprocess)\n\n\ntrain_datagen = datagen.flow_from_directory(dir,\n                                            subset = \"training\",\n                                            target_size = (MODEL_IMAGE_SIZE, MODEL_IMAGE_SIZE),\n                                            batch_size = BATCH_SIZE,\n                                            class_mode = \"categorical\")\n\n#Validation\nvalidation_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1.0/255,\n                                        validation_split=0.2,\n                                       preprocessing_function=preprocess)\n\n\nvalid_datagen = validation_datagen.flow_from_directory(dir,\n                                            subset = \"validation\",\n                                            target_size = (MODEL_IMAGE_SIZE, MODEL_IMAGE_SIZE),\n                                            batch_size = BATCH_SIZE,\n                                            class_mode = \"categorical\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 2. Defining Model (Xception)"},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -q efficientnet\nimport efficientnet.tfkeras as efn","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"inp = tf.keras.layers.Input(shape = (MODEL_IMAGE_SIZE, MODEL_IMAGE_SIZE, 3))\n\nx = efn.EfficientNetB5(weights = 'noisy-student', include_top = False)(inp)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dropout(0.2)(x)\noutput = tf.keras.layers.Dense(5, activation = 'softmax')(x)\n        \nmodel_0 = tf.keras.models.Model(inputs = [inp], outputs = [output])\n\nopt = tf.keras.optimizers.Adam(learning_rate = 0.0001)\n\nmodel_0.compile(\noptimizer = opt,\n    loss = [tf.keras.losses.CategoricalCrossentropy(label_smoothing = 0.4)],\n    metrics = [tf.keras.metrics.CategoricalAccuracy()]\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### model_0.summary()"},{"metadata":{"trusted":true},"cell_type":"code","source":"filepath = \"model_0.h5\"\n    \ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', patience=1, verbose=1, factor=0.2),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3),\n             tf.keras.callbacks.ModelCheckpoint(filepath=filepath, monitor='val_loss', save_best_only=True)]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"h = model_0.fit(train_datagen, epochs = EPOCHS, validation_data = valid_datagen, callbacks=callbacks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(h.history[\"categorical_accuracy\"], label=\"train_acc\")\nplt.plot(h.history[\"val_categorical_accuracy\"], label=\"val_acc\")\nplt.title(\"Accuracy\")\nplt.xlabel(\"Epoch \")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=\"upper left\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(h.history[\"loss\"], label=\"train_loss\")\nplt.plot(h.history[\"val_loss\"], label=\"val_loss\")\nplt.title(\"Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=\"upper left\")\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 3. Model - Cluster 1"},{"metadata":{"trusted":true},"cell_type":"code","source":"dir = './Clusters/1'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Training  Augumentation\ndatagen_1 = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1.0/255,\n                             rotation_range=30,\n                             zoom_range=0.3,\n                             horizontal_flip=True,\n                             brightness_range=[0.6, 1.2],\n                             validation_split=0.2,\n                             fill_mode='nearest',\n                             preprocessing_function=preprocess)\n\n\ntrain_datagen_1 = datagen_1.flow_from_directory(dir,\n                                            subset = \"training\",\n                                            target_size = (MODEL_IMAGE_SIZE, MODEL_IMAGE_SIZE),\n                                            batch_size = BATCH_SIZE,\n                                            class_mode = \"categorical\")\n\n#Validation\nvalidation_datagen_1 = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1.0/255,\n                                        validation_split=0.2,\n                                       preprocessing_function=preprocess)\n\n\nvalid_datagen_1 = validation_datagen_1.flow_from_directory(dir,\n                                            subset = \"validation\",\n                                            target_size = (MODEL_IMAGE_SIZE, MODEL_IMAGE_SIZE),\n                                            batch_size = BATCH_SIZE,\n                                            class_mode = \"categorical\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"inp = tf.keras.layers.Input(shape = (MODEL_IMAGE_SIZE, MODEL_IMAGE_SIZE, 3))\n\nx = efn.EfficientNetB5(weights = 'noisy-student', include_top = False)(inp)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dropout(0.2)(x)\noutput = tf.keras.layers.Dense(5, activation = 'softmax')(x)\n        \nmodel_1 = tf.keras.models.Model(inputs = [inp], outputs = [output])\n\nopt = tf.keras.optimizers.Adam(learning_rate = 0.0001)\n\nmodel_1.compile(\noptimizer = opt,\n    loss = [tf.keras.losses.CategoricalCrossentropy(label_smoothing = 0.4)],\n    metrics = [tf.keras.metrics.CategoricalAccuracy()]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filepath = \"model_1.h5\"\n    \ncallbacks = [tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', patience=1, verbose=1, factor=0.2),\n             tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3),\n             tf.keras.callbacks.ModelCheckpoint(filepath=filepath, monitor='val_loss', save_best_only=True)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"h2 = model_1.fit(train_datagen, epochs = EPOCHS, validation_data = valid_datagen, callbacks=callbacks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(h2.history[\"categorical_accuracy\"], label=\"train_acc\")\nplt.plot(h2.history[\"val_categorical_accuracy\"], label=\"val_acc\")\nplt.title(\"Accuracy\")\nplt.xlabel(\"Epoch \")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=\"upper left\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.style.use(\"ggplot\")\nplt.figure()\nplt.plot(h2.history[\"loss\"], label=\"train_loss\")\nplt.plot(h2.history[\"val_loss\"], label=\"val_loss\")\nplt.title(\"Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=\"upper left\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_0.save('model_0.tf', include_optimizer=True, save_format='tf')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_1.save('model_1.tf', include_optimizer=True, save_format='tf')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!zip -r model_0.zip 'model_0.tf'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!zip -r model_1.zip 'model_1.tf'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}