{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import needed modules","metadata":{}},{"cell_type":"code","source":"# import system libs\nimport os\nimport time\nimport shutil\nimport pathlib\nimport itertools\n\n# import data handling tools\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report\n# import Deep learning Libraries\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam, Adamax\nfrom tensorflow.keras.metrics import categorical_crossentropy\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Activation, Dropout, BatchNormalization\nfrom tensorflow.keras import regularizers\n\n# Ignore Warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nprint ('modules loaded')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create needed functions","metadata":{}},{"cell_type":"markdown","source":"**Function to split data into train, valid, test**","metadata":{}},{"cell_type":"code","source":"# Function to generate data paths with labels\ndef split_df(csv_dir):\n    '''\n    This function take csv file and split it into train, valid, and test\n    '''\n\n    df = pd.read_csv(csv_dir)\n\n    # train dataframe\n    train_df, dummy_df = train_test_split(df,  train_size= 0.7, shuffle= True, random_state= 123)\n\n    # valid and test dataframe\n    valid_df, test_df = train_test_split(dummy_df,  train_size= 0.5, shuffle= True, random_state= 123)\n\n    return train_df, valid_df, test_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to generate images from dataframe******","metadata":{}},{"cell_type":"markdown","source":"**check your variables**","metadata":{}},{"cell_type":"code","source":"def create_gens(train_df, valid_df, test_df, batch_size):\n\n    ''' This function takes train, validation, and test dataframe and fit them into image data generator, because model takes\n        data from image data generator.\n        Image data generator converts images into tensors.\n        Check your variables'''\n\n\n    # define model parameters\n    img_size = (224, 224)\n    channels = 3 # either BGR or Grayscale\n    color = 'rgb'\n    img_shape = (img_size[0], img_size[1], channels)\n    train_dir = '/kaggle/input/plant-pathology-2021-fgvc8/train_images'\n    test_dir = '/kaggle/input/plant-pathology-2021-fgvc8/test_images'\n    fpath_col = 'image'\n    label_col = 'labels'\n    tr_gen = ImageDataGenerator()\n    ts_gen = ImageDataGenerator()\n\n    train_gen = tr_gen.flow_from_dataframe( train_df, directory= train_dir, x_col= fpath_col, y_col= label_col, target_size= img_size,\n                                            class_mode= 'categorical', color_mode= color, shuffle= True, batch_size= batch_size)\n\n    valid_gen = ts_gen.flow_from_dataframe( valid_df, directory= train_dir, x_col= fpath_col, y_col= label_col, target_size= img_size,\n                                            class_mode= 'categorical', color_mode= color, shuffle= True, batch_size= batch_size)\n     # Note: we will use custom test_batch_size, and make shuffle= false\n    test_gen = ts_gen.flow_from_dataframe( test_df, directory= train_dir, x_col= fpath_col, y_col= label_col, target_size= img_size,\n                                            class_mode= 'categorical', color_mode= color, shuffle= False, batch_size= batch_size)\n\n    return train_gen, valid_gen, test_gen","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to display data sample**","metadata":{}},{"cell_type":"code","source":"def show_images(df, data_path):\n    '''\n    This function take the data generator and show sample of the images\n    '''\n\n    sample_df = df.sample(16)\n    image_names = sample_df[\"image\"].values\n    labels = sample_df[\"labels\"].values\n    plt.figure(figsize=(16, 12))\n    \n    for image_ind, (image_name, label) in enumerate(zip(image_names, labels)):\n        plt.subplot(4, 4, image_ind + 1)\n        image = cv2.imread(os.path.join(data_path, image_name))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        plt.imshow(image)\n        plt.title(f\"{label}\", fontsize=12)\n        plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to plot value counts for a column in a dataframe**","metadata":{}},{"cell_type":"code","source":"def plot_labels(df):\n    '''\n    This function take df and plot labels value counts\n    '''\n\n    plt.figure(figsize= (12, 8))\n    labels = sns.barplot(df.labels.value_counts().index,df.labels.value_counts())\n    for item in labels.get_xticklabels():\n        item.set_rotation(45)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to plot history of training**","metadata":{}},{"cell_type":"code","source":"def plot_training(hist):\n    '''\n    This function take training model and plot history of accuracy and losses with the best epoch in both of them.\n    '''\n\n    # Define needed variables\n    tr_acc = hist.history['accuracy']\n    tr_loss = hist.history['loss']\n    val_acc = hist.history['val_accuracy']\n    val_loss = hist.history['val_loss']\n    index_loss = np.argmin(val_loss)\n    val_lowest = val_loss[index_loss]\n    index_acc = np.argmax(val_acc)\n    acc_highest = val_acc[index_acc]\n    Epochs = [i+1 for i in range(len(tr_acc))]\n    loss_label = f'best epoch= {str(index_loss + 1)}'\n    acc_label = f'best epoch= {str(index_acc + 1)}'\n    # Plot training history\n    plt.figure(figsize= (20, 8))\n    plt.style.use('fivethirtyeight')\n\n    plt.subplot(1, 2, 1)\n    plt.plot(Epochs, tr_loss, 'r', label= 'Training loss')\n    plt.plot(Epochs, val_loss, 'g', label= 'Validation loss')\n    plt.scatter(index_loss + 1, val_lowest, s= 150, c= 'blue', label= loss_label)\n    plt.title('Training and Validation Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.subplot(1, 2, 2)\n    plt.plot(Epochs, tr_acc, 'r', label= 'Training Accuracy')\n    plt.plot(Epochs, val_acc, 'g', label= 'Validation Accuracy')\n    plt.scatter(index_acc + 1 , acc_highest, s= 150, c= 'blue', label= acc_label)\n    plt.title('Training and Validation Accuracy')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    plt.tight_layout\n    plt.show() ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Function to create Confusion Matrix**","metadata":{}},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes, normalize= False, title= 'Confusion Matrix', cmap= plt.cm.Blues):\n    plt.figure(figsize= (10, 10))\n    plt.imshow(cm, interpolation= 'nearest', cmap= cmap)\n    plt.title(title)\n    plt.colorbar()\n\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation= 45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis= 1)[:, np.newaxis]\n        print('Normalized Confusion Matrix')\n\n    else:\n        print('Confusion Matrix, Without Normalization')\n \n    print(cm)\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j], horizontalalignment= 'center', color= 'white' if cm[i, j] > thresh else 'black')\n\n    plt.tight_layout()\n    plt.ylabel('True Label')\n    plt.xlabel('Predicted Label')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Structure","metadata":{}},{"cell_type":"markdown","source":"**Start Reading Dataset**","metadata":{}},{"cell_type":"code","source":"csv_dir = input('Enter CSV file for training')\ntrain_path = '/kaggle/input/plant-pathology-2021-fgvc8/train_images'\n#try:\n# Get splitted data\ntrain_df, valid_df, test_df = split_df(csv_dir)\n\n# Get Generators\nbatch_size = 40\ntrain_gen, valid_gen, test_gen = create_gens(train_df, valid_df, test_df, batch_size)\n\n#except Exception:\n    #print('Invalid Input')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display Image Samples\nshow_images(train_df, train_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_labels(train_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Generic Model Creation**","metadata":{}},{"cell_type":"code","source":"# Create Model Structure\nimg_size = (224, 224)\nchannels = 3\nimg_shape = (img_size[0], img_size[1], channels)\nclass_count = len(list(train_gen.class_indices.keys())) # to define number of classes in dense layer\n\n# create pre-trained model (you can built on pretrained model such as :  efficientnet, VGG , Resnet )\n# we will use efficientnetb3 from EfficientNet family.\nbase_model = tf.keras.applications.efficientnet.EfficientNetB3(include_top= False, weights= \"imagenet\", input_shape= img_shape, pooling= 'max')\nmodel = Sequential([\n    base_model,\n    BatchNormalization(axis= -1, momentum= 0.99, epsilon= 0.001),\n    Dense(64, kernel_regularizer= regularizers.l2(l= 0.016), activity_regularizer= regularizers.l1(0.006),\n                bias_regularizer= regularizers.l1(0.006), activation= 'relu'),\n    Dropout(rate= 0.45, seed= 123),\n    Dense(class_count, activation= 'sigmoid')\n])\n\nmodel.compile(Adamax(learning_rate= 0.001), loss= 'categorical_crossentropy', metrics= ['accuracy'])\n\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train model**","metadata":{}},{"cell_type":"code","source":"history = model.fit(train_gen, epochs= epochs, verbose= 0,\n                    validation_data= valid_gen, validation_steps= None, shuffle= False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Display model performance**","metadata":{}},{"cell_type":"code","source":"plot_training(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate model","metadata":{}},{"cell_type":"code","source":"ts_length = len(test_df)\ntest_batch_size = test_batch_size = max(sorted([ts_length // n for n in range(1, ts_length + 1) if ts_length%n == 0 and ts_length/n <= 80]))\ntest_steps = ts_length // test_batch_size\n\ntrain_score = model.evaluate(train_gen, steps= test_steps, verbose= 1)\nvalid_score = model.evaluate(valid_gen, steps= test_steps, verbose= 1)\ntest_score = model.evaluate(test_gen, steps= test_steps, verbose= 1)\n\nprint(\"Train Loss: \", train_score[0])\nprint(\"Train Accuracy: \", train_score[1])\nprint('-' * 20)\nprint(\"Validation Loss: \", valid_score[0])\nprint(\"Validation Accuracy: \", valid_score[1])\nprint('-' * 20)\nprint(\"Test Loss: \", test_score[0])\nprint(\"Test Accuracy: \", test_score[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get Predictions","metadata":{}},{"cell_type":"code","source":"preds = model.predict_generator(test_gen)\ny_pred = np.argmax(preds, axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Confusion Matrics and Classification Report**","metadata":{}},{"cell_type":"code","source":"g_dict = test_gen.class_indices\nclasses = list(g_dict.keys())\n\n# Confusion matrix\ncm = confusion_matrix(test_gen.classes, y_pred)\nplot_confusion_matrix(cm= cm, classes= classes, title = 'Confusion Matrix')\n\n# Classification report\nprint(classification_report(test_gen.classes, y_pred, target_names= classes))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save model","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}