{"cells":[{"metadata":{},"cell_type":"markdown","source":"Imported libraries\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport glob\nimport random\nimport shutil\nimport warnings\nimport json\nimport itertools\nimport numpy as np\nimport pandas as pd\nfrom collections import Counter\nimport seaborn as sns\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport keras\nfrom keras.preprocessing.image import ImageDataGenerator\nimport tensorflow as tf\nfrom PIL import Image\nfrom glob import glob\n\n\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.models import Sequential\nfrom keras.layers import GlobalAveragePooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom keras.optimizers import RMSprop, Adam\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras.applications import EfficientNetB3\n\nimport os\n\nwork_dir = '../input/cassava-leaf-disease-classification/'\nos.listdir(work_dir) \ntrain_path = '/kaggle/input/cassava-leaf-disease-classification/train_images/'\n#csv_path = '/kaggle/input/cassava-leaf-disease-classification/train.csv'\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{},"cell_type":"markdown","source":"plt.figure(figsize=(8, 4))\nsns.countplot(y=\"class_name\", data=traindf);"},{"metadata":{},"cell_type":"markdown","source":"Check for GPU"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\nwith tf.device('/GPU:0'):\n    print('Yes, there is GPU')\n    \ntf.debugging.set_log_device_placement(True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Seed the dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Lets set all random seeds\n\ndef seed_everything(seed=0):\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\nseed = 66\nseed_everything(seed)\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Read the data"},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(work_dir + 'train.csv')\nprint(data['label'].value_counts()) # Checking the frequencies of the labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Importing the json file with labels\nwith open(work_dir + 'label_num_to_disease_map.json') as f:\n    real_labels = json.load(f)\n    real_labels = {int(k):v for k,v in real_labels.items()}\n    \n# Defining the working dataset\ndata['class_name'] = data['label'].map(real_labels)\n\nreal_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ndf_train[\"class_name\"] = df_train[\"label\"].map(real_labels)\n\n\ndf_train\nprint(\"The number of train images is :\", df_train.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8, 4))\nsns.countplot(y=\"class_name\", data=df_train);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_batch(image_ids, labels):\n    plt.figure(figsize=(16, 12))\n    \n    for ind, (image_id, label) in enumerate(zip(image_ids, labels)):\n        plt.subplot(4, 4, ind + 1)\n        image = cv2.imread(os.path.join(work_dir, \"train_images\", image_id))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        plt.imshow(image)\n        #plt.title(f\"Class: {label}\", fontsize=12)\n        plt.axis(\"off\")\n    \n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df = df_train.sample(16)\nimage_ids = tmp_df[\"image_id\"].values\nlabels = tmp_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Split the dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"# generate train and test sets\ntrain, test = train_test_split(data, test_size = 0.05, random_state = 42, stratify = data['class_name'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"Initialise hyperparameters"},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SIZE = 256\nsize = (IMG_SIZE,IMG_SIZE)\nn_CLASS = 5\nBATCH_SIZE = 10","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Data preprocessing"},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen_train = ImageDataGenerator(\n    #preprocessing_function = tf.keras.applications.efficientnet.preprocess_input,\n    rotation_range = 40,\n    width_shift_range = 0.2,\n    height_shift_range = 0.2,\n    shear_range = 0.2,\n    zoom_range = 0.2,\n    horizontal_flip = True,\n    vertical_flip = True,\n    fill_mode = 'nearest',\n    validation_split=0.2\n)\n\ndatagen_val = ImageDataGenerator(rescale=1/255,\n                               validation_split = 0.2\n                                )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"training & validation initialisation"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_set = datagen_train.flow_from_dataframe(\n    train,\n    directory=train_path,\n    seed=42,\n    x_col='image_id',\n    y_col='class_name',\n    target_size = size,\n    class_mode='categorical',\n    #interpolation='nearest',\n    shuffle = True,\n    batch_size = BATCH_SIZE,\n    subset = \"training\"\n)\n\ntest_set = datagen_val.flow_from_dataframe(\n    test,\n    directory=train_path,\n    seed=42,\n    x_col='image_id',\n    y_col='class_name',\n    target_size = size,\n    class_mode='categorical',\n    #interpolation='nearest',\n    shuffle=True,\n    batch_size=BATCH_SIZE,\n    subset = \"validation\"\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Building the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(\n    EfficientNetB3(\n        input_shape = (IMG_SIZE, IMG_SIZE, 3), \n        include_top = False,\n        weights='imagenet',\n        drop_connect_rate=0.6,\n    )\n)\nmodel.add(BatchNormalization(axis=-1))\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Flatten())\nmodel.add(Dense(256, activation='relu',))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(5, activation = 'softmax',))\n\nmodel.compile(loss=tf.keras.losses.CategoricalCrossentropy(), optimizer=tf.keras.optimizers.SGD(learning_rate=0.001, momentum=0.9), metrics=['acc', tf.keras.metrics.TruePositives(name='tp')])\n#model.summary()\nmodel.save('Cassava_model'+'.h5')  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def scheduler(epoch, lr):\n    if epoch >3 and epoch%2==0:\n        return lr/1.25\n    else:\n        return lr\n\n# A callback to save the model\ncallback0 = tf.keras.callbacks.ModelCheckpoint(\"./Cassava.h5\", \n                                               monitor='val_loss',save_best_only=True)\n\n# A callback to reduce the learning rate with increase in epoch\ncallback1 = tf.keras.callbacks.LearningRateScheduler(scheduler)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 5\nSTEP_SIZE_TRAIN = train_set.n // train_set.batch_size\nSTEP_SIZE_TEST = test_set.n // test_set.batch_size","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Train the model**"},{"metadata":{"trusted":true},"cell_type":"code","source":"#final_model = keras.models.load_model('Cassava_model.h5') \nresults = model.fit(train_set, validation_data=test_set, epochs=5, callbacks=[callback0, callback1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"draw the results & try to fix the bug of an empty plot"},{"metadata":{"trusted":true},"cell_type":"code","source":"import plotly.graph_objects as go\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"\ndef trai_test_plot(acc, test_acc, loss, test_loss):\n    \n    fig, (ax1, ax2) = plt.subplots(1,2, figsize= (15,10))\n    fig.suptitle(\"Model's metrics comparisson\", fontsize=20)\n\n    ax1.plot(range(1, len(acc) + 1), acc)\n    ax1.plot(range(1, len(test_acc) + 1), test_acc)\n    ax1.set_title('History of Accuracy', fontsize=15)\n    ax1.set_xlabel('Epochs', fontsize=15)\n    ax1.set_ylabel('Accuracy', fontsize=15)\n    ax1.legend(['training', 'validation'])\n\n\n    ax2.plot(range(1, len(loss) + 1), loss)\n    ax2.plot(range(1, len(test_loss) + 1), test_loss)\n    ax2.set_title('History of Loss', fontsize=15)\n    ax2.set_xlabel('Epochs', fontsize=15)\n    ax2.set_ylabel('Loss', fontsize=15)\n    ax2.legend(['training', 'validation'])\n    plt.show()\n    \n\ntrai_test_plot(\n    results.history['categorical_accuracy'],\n    results.history['val_categorical_accuracy'],\n    results.history['loss'],\n    results.history['val_loss']\n)"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Predictions**"},{"metadata":{"trusted":true},"cell_type":"code","source":"STEP_SIZE = test_set.n // test_set.batch_size\nfinal_model = keras.models.load_model('Cassava_model.h5')\npredict = final_model.predict(test_set, STEP_SIZE)\nprint(predict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('here')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = []\nss = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\n\nfor image in ss.image_id:\n    img = tf.keras.preprocessing.image.load_img('../input/cassava-leaf-disease-classification/test_images/' + image)\n    img = tf.keras.preprocessing.image.img_to_array(img)\n    img = tf.keras.preprocessing.image.smart_resize(img, (IMG_SIZE, IMG_SIZE))\n    img = tf.reshape(img, (-1, IMG_SIZE, IMG_SIZE, 3))\n    prediction = model.predict(img/255)\n    preds.append(np.argmax(prediction))\n\nmy_submission = pd.DataFrame({'image_id': ss.image_id, 'label': preds})\nmy_submission.to_csv('submission.csv', index=False) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Submission File: \\n---------------\\n\")\nprint(submission.head()) ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**EVALUATION**"},{"metadata":{"trusted":true},"cell_type":"code","source":"STEP_SIZE = test_set.n // test_set.batch_size\n\npredict = model.predict(test_set, STEP_SIZE)\nprint(predict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_gen = ImageDataGenerator().flow_from_dataframe(\n    submission,\n    directory=\"/kaggle/input/landmark-recognition-2020/test/\",\n    x_col=\"image_id\",\n    y_col=\"label\",\n    weight_col=None,\n    target_size=(IMG_SIZE, IMG_SIZE),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=None,\n    batch_size=1,\n    shuffle=True,\n    subset=None,\n    interpolation=\"nearest\",\n    validate_filenames=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(list(label_map))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"good_preds = []\nbad_preds = []\nval_filenames = test_set.filenames\nlabel_map = (test_set.class_indices)\n#label_categories = to_categorical(np.asarray(labels)) \ncla = np.argmax(predict, axis=-1)\nlabel_map = list(map(int, label_map.values()))\nval_label = test_set.labels\n\nfor idx, res in enumerate(predict):\n    #print(\"image_id: \", val_filenames[idx], \", class predict: \", label_map[cla[idx]], \"class: \", label_map[val_label[idx]])\n    \n    if label_map[cla[idx]] != label_map[val_label[idx]]:\n        bad_preds.append([val_filenames[idx], label_map[cla[idx]], label_map[val_label[idx]], res[cla[idx]]])\n    else:\n        good_preds.append([val_filenames[idx], label_map[cla[idx]], label_map[val_label[idx]], res[cla[idx]]])\nprint(\"wrong predictions: \", len(bad_preds), \" right predictions: \", len(good_preds), \" acc: \", np.round(100*(len(predict)-len(bad_preds))/len(predict),2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=plt.figure(figsize=(16, 8))\n\ngood_preds = np.array(good_preds)\ngood_preds = np.array(sorted(good_preds, key = lambda x: x[3], reverse=True))\n#print(good_preds.shape)\n\ncolumns = 5\nrows = 1\nfor i in range(1, columns*rows +1):\n    n = good_preds[i,0]\n    #print(n)\n    img = cv2.imread(os.path.join(train_path,n))\n    lbl = good_preds[i,2]\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img)\n    lbl2 = good_preds[i,1]\n    plt.title(\"Label = \" + str(lbl) + \"\\nClassified:\" + str(lbl2) + \"\\nConfidence:\" + str(good_preds[i,3]))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### plot the worst predictions\n\nfig=plt.figure(figsize=(16, 8))\n\nbad_preds = np.array(bad_preds)\nbad_preds = np.array(sorted(bad_preds, key = lambda x: x[3], reverse=True))\n#print(bad_preds.shape)\n\ncolumns = 5\nrows = 1\nfor i in range(1, columns*rows +1):\n    n = bad_preds[i,0]\n    #print(n)\n    img = cv2.imread(os.path.join(train_path,n))\n    lbl = bad_preds[i,2]\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img)\n    lbl2 = bad_preds[i,1]\n    plt.title(\"Label = \" + str(lbl) + \"\\nClassified:\" + str(lbl2) + \"\\nConfidence:\" + str(good_preds[i,3]))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}