{"cells":[{"metadata":{},"cell_type":"markdown","source":"Introduction\n\nIn this competition the task is to classify each cassava image into multiple categories to indicate if the leaf is healty or not. \n\nTo do this, first a small EDA will be written to see what the data looks like. "},{"metadata":{},"cell_type":"markdown","source":"Here are the necessary python libraries that are needed for the EDA"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras import datasets, layers, models\nfrom keras.models import Sequential\nfrom fastai.basics import *\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Here the different files will be listed for working with the cassava leaf detection competition"},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndataset = Path('../input/cassava-leaf-disease-classification')\nos.listdir(dataset)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"With pandas the train.csv will be readed and the first 5 pictures will be listed in a tabel"},{"metadata":{"trusted":true},"cell_type":"code","source":"trainset = pd.read_csv(dataset / 'train.csv')\ntrainset.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"With dtypes you can see the different data types in the csv file"},{"metadata":{"trusted":true},"cell_type":"code","source":"trainset.dtypes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Below are the unique values in the dataset. There are 21397 unique values"},{"metadata":{"trusted":true},"cell_type":"code","source":"trainset.image_id.nunique()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Here you can see how many values are in the different labels. Label 3 has the most unique values"},{"metadata":{"trusted":true},"cell_type":"code","source":"trainset['label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"With isnull() you can see how many 0 values are in the dataset. Below are the results"},{"metadata":{"trusted":true},"cell_type":"code","source":"trainset.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"### with open(os.path.join(dataset, \"label_num_to_disease_map.json\")) as file:\n    map_classes = json.loads(file.read())\n    classes=[]\n    for k, v in map_classes.items():        \n        classes.append([int(k), v])\n        \nclassesframe = pd.DataFrame(classes, columns=['Label', 'Label Details'])\nprint(classesframe.to_string(index=False))"},{"metadata":{},"cell_type":"markdown","source":"Because in the other file the label names are stored, below is the formula to get the names and the count for every label."},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = ['CBB', 'CBSD', 'CGM', 'CMD', 'Healthy']\nax = sns.countplot(x=\"label\", data=trainset)\nfor p in ax.patches:\n    ax.annotate(format(p.get_height(), '.1f'), \n                   (p.get_x() + p.get_width() / 2., p.get_height()), \n                   ha = 'center', va = 'center', \n                   xytext = (0, 5), \n                   textcoords = 'offset points')\nax.set_xlabel('Label details')\nax.set_ylabel('Frequency')\nax.set_xticklabels(labels)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainlabel_0 = trainset[trainset['label'] == 0]\ntrainlabel_1 = trainset[trainset['label'] == 1]\ntrainlabel_2 = trainset[trainset['label'] == 2]\ntrainlabel_3 = trainset[trainset['label'] == 3]\ntrainlabel_4 = trainset[trainset['label'] == 4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"picture_train_path = Path('../input/cassava-leaf-disease-classification/train_images')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\nimage_0 = Image.open(picture_train_path/trainlabel_0.loc[trainlabel_0.index[0]]['image_id'])\nimage_1 = Image.open(picture_train_path/trainlabel_1.loc[trainlabel_1.index[0]]['image_id'])\nimage_2 = Image.open(picture_train_path/trainlabel_2.loc[trainlabel_2.index[0]]['image_id'])\nimage_3 = Image.open(picture_train_path/trainlabel_3.loc[trainlabel_3.index[0]]['image_id'])\nimage_4 = Image.open(picture_train_path/trainlabel_4.loc[trainlabel_4.index[0]]['image_id'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_4","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# thanks to this kaggle notebook https://www.kaggle.com/kimse0ha/cassava-leaf-disease-classification-eda \n\nimport json\nimport matplotlib as mpl\n\n\nwith open('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json', 'r') as f:\n    json_data = json.load(f)\n\nbase_dir =(dataset / 'train_images/')\nfor i in range(5):\n    globals()['label_{}'.format(i)] = trainset[trainset.label==i].image_id.tolist()\n    plt.figure(figsize=(20, 15))\n    plt.suptitle('Label {} : {}'.format(i, json_data['{}'.format(i)]), fontsize=25)\n    \n    for i, filename in enumerate(np.random.choice(globals()['label_{}'.format(i)], 25)):\n        plt.subplot(5, 5, i+1)\n        plt.axis('off')\n        img = mpl.image.imread(os.path.join(base_dir, filename))\n        plt.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\n\ntransposed  = image_1.transpose(Image.ROTATE_90)\ntransposed.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"targetsize = 64\nbatchsize = 64\nsteps = len(trainset)*0.7 / batchsize\nvalidation = len(trainset)*0.3 / batchsize\nepochs = 4","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainset.label = trainset.label.astype('str')\n\ntrain_generator = ImageDataGenerator(validation_split = 0.2,\n                                     preprocessing_function = None,\n                                     zoom_range = 0.15,\n                                     cval = 0.,\n                                     horizontal_flip = True,\n                                     vertical_flip = True,\n                                     fill_mode = 'nearest',\n                                     shear_range = 0.15,\n                                     height_shift_range = 0.15,\n                                     width_shift_range = 0.15) \\\n    .flow_from_dataframe(trainset,\n                         directory = os.path.join(dataset, \"train_images\"),\n                         subset = \"training\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (targetsize, targetsize),\n                         batch_size = batchsize,\n                         class_mode = \"sparse\")\n\nvalidation_generator = ImageDataGenerator(validation_split = 0.2) \\\n    .flow_from_dataframe(trainset,\n                         directory = os.path.join(dataset, \"train_images\"),\n                         subset = \"validation\",\n                         x_col = \"image_id\",\n                         y_col = \"label\",\n                         target_size = (targetsize, targetsize),\n                         batch_size = batchsize,\n                         class_mode = \"sparse\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def createCNNModel() :\n    model = Sequential()\n\n    model.add(Conv2D(16, (3, 3), activation='relu', input_shape=(targetsize, targetsize, 3)))\n    model.add(MaxPooling2D((2, 2)))\n    \n    model.add(Conv2D(32, (3,3), activation='relu'))\n    model.add(MaxPooling2D((2, 2)))\n    \n    model.add(Conv2D(64, (3, 3), activation='relu'))\n    model.add(MaxPooling2D((2, 2)))\n    \n    model.add(Conv2D(64, (3, 3), activation='relu'))\n    model.add(Flatten())\n    model.add(Dense(128, activation='relu'))\n    model.add(Dense(256, activation='relu'))\n    model.add(Dense(5, activation='softmax'))\n    \n    model.compile(optimizer = 'Adam',\n                  loss = \"sparse_categorical_crossentropy\",\n                  metrics = [\"acc\"])\n   \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = createCNNModel()\n\nmodel.summary()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(model.layers)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keras.utils.plot_model(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    steps_per_epoch = steps,\n    epochs = 4,\n    validation_data = validation_generator,\n    validation_steps = validation,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_dict = history.history\nprint(history_dict.keys())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['acc'])\nplt.plot(history.history['val_acc'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.layers.normalization import BatchNormalization\n\ndef secondCNNmodel() :\n    res = (224, 224)\n    model = Sequential()\n    \n    model.add(Conv2D(16, (3, 3), activation='relu', input_shape=(targetsize, targetsize, 3)))\n    model.add(Flatten())\n    \n    model.add(BatchNormalization())\n    model.add(Dense(256, activation='relu'))\n    model.add(Dropout(0.5))\n    \n    model.add(BatchNormalization())\n    model.add(Dense(128, activation='relu'))\n    model.add(Dropout(0.5))\n    \n    model.add(BatchNormalization())\n    model.add(Dense(64, activation='relu'))\n    model.add(Dropout(0.5))\n    \n    model.add(BatchNormalization())\n    model.add(Dense(10, activation='softmax'))\n    \n    model.compile(optimizer = 'Adam',\n                  loss = \"sparse_categorical_crossentropy\",\n                  metrics = [\"acc\"])\n   \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model2 = secondCNNmodel()\nmodel2.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(model2.layers)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keras.utils.plot_model(model2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history2 = model2.fit(\n    train_generator,\n    steps_per_epoch = steps,\n    epochs = epochs,\n    validation_data = validation_generator,\n    validation_steps = validation,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_dict2 = history2.history\nprint(history_dict2.keys())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history2.history['acc'])\nplt.plot(history2.history['val_acc'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history2.history['loss'])\nplt.plot(history2.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Transfer learning epochs visualisatie"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications import ResNet50\nmodel3 = ResNet50()\nmodel3.compile(optimizer = 'Adam',\n                  loss = \"sparse_categorical_crossentropy\",\n                  metrics = [\"acc\"])\nmodel3.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(model3.layers)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keras.utils.plot_model(model3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history3 = model3.fit(\n    train_generator,\n    steps_per_epoch = 3,\n    epochs = epochs,\n    validation_data = validation_generator,\n    validation_steps = validation,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_dict3 = history3.history\nprint(history_dict3.keys())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history3.history['acc'])\nplt.plot(history3.history['val_acc'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history3.history['loss'])\nplt.plot(history3.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for param in model3.parameters():\n    param.requires_grad = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch.nn as nn\n\nmodel3.classifier[6] = nn.Sequential(\n                      nn.Linear(n_inputs, 256), \n                      nn.ReLU(), \n                      nn.Dropout(0.4),\n                      nn.Linear(256, n_classes),                   \n                      nn.LogSoftmax(dim=1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Find total parameters and trainable parameters\ntotal_params = sum(p.numel() for p in model3.parameters())\nprint(f'{total_params:,} total parameters.')\ntotal_trainable_params = sum(\n    p.numel() for p in model3.parameters() if p.requires_grad)\nprint(f'{total_trainable_params:,} training parameters.')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications.vgg16 import VGG16\nfrom keras.applications.vgg16 import preprocess_input\nfrom keras.preprocessing.image import load_img\nfrom keras.preprocessing.image import img_to_array\nfrom keras.models import Model\nfrom matplotlib import pyplot\n\nmodel = VGG16()\nmodel = Model(inputs=model.inputs, outputs=model.layers[1].output)\nmodel.summary()\nimg = load_img(image_1, target_size=(224, 224))\nimg = img_to_array(img)\nimg = expand_dims(img, axis=0)\nimg = preprocess_input(img)\nfeature_maps = model.predict(img)\nsquare = 8\nix = 1\nfor _ in range(square):\n\tfor _ in range(square):\n\t\t# specify subplot and turn of axis\n\t\tax = pyplot.subplot(square, square, ix)\n\t\tax.set_xticks([])\n\t\tax.set_yticks([])\n\t\t# plot filter channel in grayscale\n\t\tpyplot.imshow(feature_maps[0, :, :, ix-1], cmap='gray')\n\t\tix += 1\npyplot.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Sources that are used for making this notebook\nhttps://nanonets.com/blog/data-augmentation-how-to-use-deep-learning-when-you-have-limited-data-part-2/\nhttps://medium.com/@kenneth.ca95/a-guide-to-transfer-learning-with-keras-using-resnet50-a81a4a28084b\nhttps://www.pluralsight.com/guides/data-visualization-deep-learning-model-using-matplotlib"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}