{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n<h1 style='background:#2cab6c; border:0; color:white'><center>Importing Libraries</center></h1>","metadata":{}},{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport cv2 as cv\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport plotly.graph_objects as go\n\nfrom pathlib import Path\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:04.627313Z","iopub.execute_input":"2022-01-12T10:15:04.627698Z","iopub.status.idle":"2022-01-12T10:15:08.557088Z","shell.execute_reply.started":"2022-01-12T10:15:04.627604Z","shell.execute_reply":"2022-01-12T10:15:08.556116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Paths, files</center></h1>","metadata":{}},{"cell_type":"code","source":"# Paths to the base directories/files of the dataset\nbase_dir = Path('/kaggle/input/cassava-leaf-disease-classification')\ntrain_img_dir = f'{base_dir}/train_images'\ntest_img_dir = f'{base_dir}/test_images'","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:11.448002Z","iopub.execute_input":"2022-01-12T10:15:11.448289Z","iopub.status.idle":"2022-01-12T10:15:11.453364Z","shell.execute_reply.started":"2022-01-12T10:15:11.448257Z","shell.execute_reply":"2022-01-12T10:15:11.452131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read train csv and json files with labels mapped to disease names\ntrain_df = pd.read_csv(f'{base_dir}/train.csv')\ndisease_mapping = pd.read_json(f'{base_dir}/label_num_to_disease_map.json', typ='series')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:12.723339Z","iopub.execute_input":"2022-01-12T10:15:12.724018Z","iopub.status.idle":"2022-01-12T10:15:12.771087Z","shell.execute_reply.started":"2022-01-12T10:15:12.723981Z","shell.execute_reply":"2022-01-12T10:15:12.770086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create lists with all train and test images\ntrain_images = os.listdir(f'{base_dir}/train_images/')\ntest_images = os.listdir(f'{base_dir}/test_images/')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:13.846889Z","iopub.execute_input":"2022-01-12T10:15:13.850026Z","iopub.status.idle":"2022-01-12T10:15:14.360467Z","shell.execute_reply.started":"2022-01-12T10:15:13.849983Z","shell.execute_reply":"2022-01-12T10:15:14.359158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Labels Mapping, Training Dataset</center></h1>","metadata":{}},{"cell_type":"code","source":"# Convert mapping to dict\nmapping_dict = disease_mapping.to_dict()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:20.723678Z","iopub.execute_input":"2022-01-12T10:15:20.723963Z","iopub.status.idle":"2022-01-12T10:15:20.728497Z","shell.execute_reply.started":"2022-01-12T10:15:20.723932Z","shell.execute_reply":"2022-01-12T10:15:20.727480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show dict\nmapping_dict","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:20.975009Z","iopub.execute_input":"2022-01-12T10:15:20.975603Z","iopub.status.idle":"2022-01-12T10:15:20.985642Z","shell.execute_reply.started":"2022-01-12T10:15:20.975569Z","shell.execute_reply":"2022-01-12T10:15:20.984477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, the dataset contains 5 classes","metadata":{}},{"cell_type":"code","source":"# Show first 10 lines of train dataset\ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:25.493884Z","iopub.execute_input":"2022-01-12T10:15:25.494198Z","iopub.status.idle":"2022-01-12T10:15:25.513351Z","shell.execute_reply.started":"2022-01-12T10:15:25.494151Z","shell.execute_reply":"2022-01-12T10:15:25.512375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the name of the images and their class labels here. The labels are represented as numbers, but for convenience we can replace them with the appropriate names.","metadata":{}},{"cell_type":"code","source":"# Let's check for any missing values in the train labels\nmissing = train_df.isnull().sum()\nall_value = train_df.count()\n\nmissing_df = pd.concat([missing, all_value], axis=1, keys=['Missing Val.', 'All Val.'])\nmissing_df","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:27.890466Z","iopub.execute_input":"2022-01-12T10:15:27.890816Z","iopub.status.idle":"2022-01-12T10:15:27.918460Z","shell.execute_reply.started":"2022-01-12T10:15:27.890784Z","shell.execute_reply":"2022-01-12T10:15:27.917424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see that there are no missing in the data.","metadata":{}},{"cell_type":"code","source":"# Let's replace numeric labels in dataset with disease names\ntrain_df = train_df.replace(mapping_dict)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:28.545494Z","iopub.execute_input":"2022-01-12T10:15:28.546119Z","iopub.status.idle":"2022-01-12T10:15:28.565377Z","shell.execute_reply.started":"2022-01-12T10:15:28.546074Z","shell.execute_reply":"2022-01-12T10:15:28.564453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show first 10 lines of train dataset with replaced labels\ntrain_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:29.985964Z","iopub.execute_input":"2022-01-12T10:15:29.986271Z","iopub.status.idle":"2022-01-12T10:15:29.998590Z","shell.execute_reply.started":"2022-01-12T10:15:29.986241Z","shell.execute_reply":"2022-01-12T10:15:29.997480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's count the num of training samples for each label\nlabel_counts = train_df['label'].value_counts().reset_index()\nlabel_counts.columns = ['Label', 'Num. of Observations']\n\n# Create Pie Chart\nfig = px.pie(label_counts,\n             names='Label', values='Num. of Observations',\n             labels=mapping_dict,\n             title='Percentage Distribution of Labels in the Training Dataset')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:32.586479Z","iopub.execute_input":"2022-01-12T10:15:32.587094Z","iopub.status.idle":"2022-01-12T10:15:33.755495Z","shell.execute_reply.started":"2022-01-12T10:15:32.587058Z","shell.execute_reply":"2022-01-12T10:15:33.754558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see that the dataset contains a significant class imbalance, where most of the images are of the Cassava Mosaic Disease (CMD) class.\n\nThe dataset contains only 12% of the data with images of healthy leaves, while all other images are for diseased leaves.","metadata":{}},{"cell_type":"code","source":"# Let's check if the dataset contains duplicate images\nunique_idx = train_df['image_id'].nunique()\n\nif unique_idx == len(train_df):\n    print('There are no duplicate image indices in the training dataset.')\nelse:\n    print(f'There are {len(train_df) - unique_idx} duplicate image indices in the training dataset.')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:37.462196Z","iopub.execute_input":"2022-01-12T10:15:37.462491Z","iopub.status.idle":"2022-01-12T10:15:37.477704Z","shell.execute_reply.started":"2022-01-12T10:15:37.462459Z","shell.execute_reply":"2022-01-12T10:15:37.476097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Training Image Dataset</center></h1>","metadata":{"execution":{"iopub.status.busy":"2022-01-10T10:55:32.864661Z","iopub.execute_input":"2022-01-10T10:55:32.864929Z","iopub.status.idle":"2022-01-10T10:55:32.870227Z","shell.execute_reply.started":"2022-01-10T10:55:32.864897Z","shell.execute_reply":"2022-01-10T10:55:32.869357Z"}}},{"cell_type":"code","source":"# Let's check how many images are in the training dataset\nprint(f'{len(train_images)} training images contains dataset.')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:41.798194Z","iopub.execute_input":"2022-01-12T10:15:41.798719Z","iopub.status.idle":"2022-01-12T10:15:41.804499Z","shell.execute_reply.started":"2022-01-12T10:15:41.798678Z","shell.execute_reply":"2022-01-12T10:15:41.803451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's check the resolution of the images to make sure they're all standardized\nimgs_shape = []\n\nfor img in tqdm(train_images):\n    image = cv.imread(f'{base_dir}/train_images/{img}')\n    imgs_shape.append(image.shape)\n    \nprint(f'The training set contains the following unique image shapes: {set(imgs_shape)}')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:15:42.075246Z","iopub.execute_input":"2022-01-12T10:15:42.075856Z","iopub.status.idle":"2022-01-12T10:21:29.364381Z","shell.execute_reply.started":"2022-01-12T10:15:42.075820Z","shell.execute_reply":"2022-01-12T10:21:29.363191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that all the images have the same shape: 600 by 800 pixels with 3 channels.","metadata":{}},{"cell_type":"code","source":"# Let's break the training dataset into separate lists with images for each class\nhealthy_imgs = train_df[train_df['label'] == 'Healthy']['image_id'].to_list()\ncmd_imgs = train_df[train_df['label'] == 'Cassava Mosaic Disease (CMD)']['image_id'].to_list()\ncgm_imgs = train_df[train_df['label'] == 'Cassava Green Mottle (CGM)']['image_id'].to_list()\ncbsd_imgs = train_df[train_df['label'] == 'Cassava Brown Streak Disease (CBSD)']['image_id'].to_list()\ncbb = train_df[train_df['label'] == 'Cassava Bacterial Blight (CBB)']['image_id'].to_list()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:25.455423Z","iopub.execute_input":"2022-01-12T10:22:25.455796Z","iopub.status.idle":"2022-01-12T10:22:25.484575Z","shell.execute_reply.started":"2022-01-12T10:22:25.455748Z","shell.execute_reply":"2022-01-12T10:22:25.483684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, we will implement a couple of helper functions for displaying images and their color histograms.","metadata":{}},{"cell_type":"code","source":"def show_img(imgs_list, title):\n    \"\"\"Function for displaying images\n    \n    Args:\n        img_list (list): a list that contains the names of the image files\n        title (str): class label name\n    \"\"\"\n    images = [np.random.choice(imgs_list) for i in range(6)]\n    \n    plt.figure(figsize=(12, 12))\n    plt.suptitle(title, fontsize=24)\n    \n    for i in range(6):\n        plt.subplot(3, 3, i+1)\n        img = plt.imread(f'{train_img_dir}/{images[i]}')\n        plt.imshow(img, cmap='gray')\n        plt.axis('off')\n    \n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:32.954397Z","iopub.execute_input":"2022-01-12T10:22:32.955270Z","iopub.status.idle":"2022-01-12T10:22:32.964666Z","shell.execute_reply.started":"2022-01-12T10:22:32.955211Z","shell.execute_reply":"2022-01-12T10:22:32.961809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_hist(imgs_list, title):\n    \"\"\"Function to display a random image from a dataset and its histogram of color channels\n    \n    Args:\n        img_list (list): a list that contains the names of the image files\n        title (str): class label name\n    \"\"\"\n    image = np.random.choice(imgs_list)\n    \n    fig = plt.figure(figsize=(12, 12))\n    \n    fig.add_subplot(1, 2, 1)\n    \n    img = plt.imread(f'{train_img_dir}/{image}')\n    \n    plt.title(title)\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n    print(f'Image dimensions: {img.shape[0], img.shape[1]}',\n          f'Max pixel value: {img.max()}',\n          f'Min pixel value: {img.min()}',\n          f'Mean pixel value: {round(img.mean())}',\n          f'Standard deviation: {round(img.std())}', sep='\\n')\n    \n    fig.add_subplot(1, 2, 2)\n    \n    plt.hist(img[:, :,  0].ravel(), bins=256, color='red', alpha=0.5)\n    plt.hist(img[:, :,  1].ravel(), bins=256, color='green', alpha=0.5)\n    plt.hist(img[:, :,  2].ravel(), bins=256, color='blue', alpha=0.5)\n    \n    plt.xlabel('Intensity Value')\n    plt.ylabel('Count')\n    plt.legend(['Red Channel', 'Green Channel', 'Blue Channel'])\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:33.176327Z","iopub.execute_input":"2022-01-12T10:22:33.176694Z","iopub.status.idle":"2022-01-12T10:22:33.188286Z","shell.execute_reply.started":"2022-01-12T10:22:33.176662Z","shell.execute_reply":"2022-01-12T10:22:33.187038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rgb_image(image_idx):\n    \"\"\"Function for getting a NumPy Array of an image and converting it from BGR to RGB\n    \n    Args:\n        image_idx (str): image file name\n        \n    Returns:\n        ndarray: NumPy Array of an image\n    \"\"\"\n    img = cv.imread(f'{train_img_dir}/{image_idx}')\n    img = cv.cvtColor(img, cv.COLOR_BGR2RGB)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:34.330493Z","iopub.execute_input":"2022-01-12T10:22:34.330829Z","iopub.status.idle":"2022-01-12T10:22:34.338666Z","shell.execute_reply.started":"2022-01-12T10:22:34.330788Z","shell.execute_reply":"2022-01-12T10:22:34.336350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_histograms_data(imgs_list):\n    \"\"\"Function for getting median values of color channels of input images\n    \n    Args:\n        img_list (list): a list that contains the names of the image files\n        \n    Returns:\n        list: a list that contains the median values of the color channels of images\n    \"\"\"\n    img = [get_rgb_image(image_idx) for image_idx in imgs_list]\n    \n    red_values = [np.mean(img[idx][:, :, 0]) for idx in range(len(img))]\n    green_values = [np.mean(img[idx][:, :, 1]) for idx in range(len(img))]\n    blue_values = [np.mean(img[idx][:, :, 2]) for idx in range(len(img))]\n    all_mean_values = [np.mean(img[idx]) for idx in range(len(img))]\n    \n    return [red_values, green_values, blue_values, all_mean_values]","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:34.687960Z","iopub.execute_input":"2022-01-12T10:22:34.688876Z","iopub.status.idle":"2022-01-12T10:22:34.697637Z","shell.execute_reply.started":"2022-01-12T10:22:34.688834Z","shell.execute_reply":"2022-01-12T10:22:34.696338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_box_plot(imgs_list, title):\n    \"\"\"Function for displaying box-plots of histogram of color channels of images\n    \n    Args:\n        img_list (list): a list that contains the names of the image files\n        title (str): class label name\n    \"\"\"\n    fig_data = []\n    hist_data = get_histograms_data(imgs_list)\n    \n    for i, name in zip(range(3), ['Red', 'Green', 'Blue']):\n        mark = go.Box(y=hist_data[i],\n                      name=name,\n                      boxpoints='all',\n                      marker_color=name)\n        fig_data.append(mark)\n    \n    fig = go.Figure(fig_data)\n    fig.update_layout(title_text=f'{title} - Distribution of Pixel Values')\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:35.967169Z","iopub.execute_input":"2022-01-12T10:22:35.967622Z","iopub.status.idle":"2022-01-12T10:22:35.975348Z","shell.execute_reply.started":"2022-01-12T10:22:35.967585Z","shell.execute_reply":"2022-01-12T10:22:35.974113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Test Image Dataset</center></h1>","metadata":{"execution":{"iopub.status.busy":"2022-01-10T11:01:31.21157Z","iopub.execute_input":"2022-01-10T11:01:31.212301Z","iopub.status.idle":"2022-01-10T11:01:31.21697Z","shell.execute_reply.started":"2022-01-10T11:01:31.212252Z","shell.execute_reply":"2022-01-10T11:01:31.215809Z"}}},{"cell_type":"code","source":"# Let's check how many images the test dataset contains\nprint(f'The training set contains {len(test_images)} image.')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:22:53.470071Z","iopub.execute_input":"2022-01-12T10:22:53.470371Z","iopub.status.idle":"2022-01-12T10:22:53.476655Z","shell.execute_reply.started":"2022-01-12T10:22:53.470338Z","shell.execute_reply":"2022-01-12T10:22:53.475613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, the test case contains only one image. Therefore, when training the model, we will need to split the training dataset into test and validation.\n\nAlso, as stated in the description of the Kaggle competition, the test set contains 15k images, and it becomes available only after we send our notebook for scoring.","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Healthy</center></h1>","metadata":{}},{"cell_type":"code","source":"show_img(healthy_imgs, 'Healthy')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:23:00.838570Z","iopub.execute_input":"2022-01-12T10:23:00.838980Z","iopub.status.idle":"2022-01-12T10:23:02.058963Z","shell.execute_reply.started":"2022-01-12T10:23:00.838933Z","shell.execute_reply":"2022-01-12T10:23:02.056743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_hist(healthy_imgs, 'Healthy')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:23:02.060539Z","iopub.execute_input":"2022-01-12T10:23:02.061569Z","iopub.status.idle":"2022-01-12T10:23:04.858397Z","shell.execute_reply.started":"2022-01-12T10:23:02.061524Z","shell.execute_reply":"2022-01-12T10:23:04.857455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_box_plot(healthy_imgs, 'Healthy')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:23:04.860404Z","iopub.execute_input":"2022-01-12T10:23:04.861527Z","iopub.status.idle":"2022-01-12T10:23:38.155501Z","shell.execute_reply.started":"2022-01-12T10:23:04.861466Z","shell.execute_reply":"2022-01-12T10:23:38.154326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the following distribution of median values for pixel intensities:\n\n- Red: 108\n- Green: 126\n- Blue: 80","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Cassava Mosaic Disease (CMD)</center></h1>","metadata":{}},{"cell_type":"code","source":"show_img(cmd_imgs, 'Cassava Mosaic Disease (CMD)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:23:38.157787Z","iopub.execute_input":"2022-01-12T10:23:38.158228Z","iopub.status.idle":"2022-01-12T10:23:39.506890Z","shell.execute_reply.started":"2022-01-12T10:23:38.158168Z","shell.execute_reply":"2022-01-12T10:23:39.505820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_hist(cmd_imgs, 'Cassava Mosaic Disease (CMD)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:23:39.508518Z","iopub.execute_input":"2022-01-12T10:23:39.509027Z","iopub.status.idle":"2022-01-12T10:23:42.616256Z","shell.execute_reply.started":"2022-01-12T10:23:39.508972Z","shell.execute_reply":"2022-01-12T10:23:42.615274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_box_plot(cmd_imgs[:5000], 'Cassava Mosaic Disease (CMD)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:23:42.619057Z","iopub.execute_input":"2022-01-12T10:23:42.619655Z","iopub.status.idle":"2022-01-12T10:24:48.606158Z","shell.execute_reply.started":"2022-01-12T10:23:42.619613Z","shell.execute_reply":"2022-01-12T10:24:48.605280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For this class, we only used the first 5000 images.\n\nWe can see the following distribution of median values for pixel intensities:\n\n- Red: 109\n- Green: 128\n- Blue: 79","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Cassava Green Mottle (CGM)</center></h1>","metadata":{}},{"cell_type":"code","source":"show_img(cgm_imgs, 'Cassava Green Mottle (CGM)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:24:48.607852Z","iopub.execute_input":"2022-01-12T10:24:48.608412Z","iopub.status.idle":"2022-01-12T10:24:49.737884Z","shell.execute_reply.started":"2022-01-12T10:24:48.608345Z","shell.execute_reply":"2022-01-12T10:24:49.736773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_hist(cgm_imgs, 'Cassava Green Mottle (CGM)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:24:49.739548Z","iopub.execute_input":"2022-01-12T10:24:49.740077Z","iopub.status.idle":"2022-01-12T10:24:53.411542Z","shell.execute_reply.started":"2022-01-12T10:24:49.740036Z","shell.execute_reply":"2022-01-12T10:24:53.410703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_box_plot(cgm_imgs, 'Cassava Green Mottle (CGM)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:24:53.413193Z","iopub.execute_input":"2022-01-12T10:24:53.413676Z","iopub.status.idle":"2022-01-12T10:25:24.133987Z","shell.execute_reply.started":"2022-01-12T10:24:53.413623Z","shell.execute_reply":"2022-01-12T10:25:24.133051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the following distribution of median values for pixel intensities:\n\n- Red: 113\n- Green: 128\n- Blue: 85","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Cassava Brown Streak Disease (CBSD)</center></h1>","metadata":{}},{"cell_type":"code","source":"show_img(cbsd_imgs, 'Cassava Brown Streak Disease (CBSD)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:25:24.135560Z","iopub.execute_input":"2022-01-12T10:25:24.136043Z","iopub.status.idle":"2022-01-12T10:25:25.325920Z","shell.execute_reply.started":"2022-01-12T10:25:24.136003Z","shell.execute_reply":"2022-01-12T10:25:25.323152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_hist(cbsd_imgs, 'Cassava Brown Streak Disease (CBSD)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:25:25.327674Z","iopub.execute_input":"2022-01-12T10:25:25.328234Z","iopub.status.idle":"2022-01-12T10:25:28.299671Z","shell.execute_reply.started":"2022-01-12T10:25:25.328179Z","shell.execute_reply":"2022-01-12T10:25:28.298801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_box_plot(cbsd_imgs, 'Cassava Brown Streak Disease (CBSD)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:25:28.302510Z","iopub.execute_input":"2022-01-12T10:25:28.303578Z","iopub.status.idle":"2022-01-12T10:25:56.700349Z","shell.execute_reply.started":"2022-01-12T10:25:28.303530Z","shell.execute_reply":"2022-01-12T10:25:56.699442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the following distribution of median values for pixel intensities:\n\n- Red: 106\n- Green: 123\n- Blue: 72","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Cassava Bacterial Blight (CBB)</center></h1>","metadata":{}},{"cell_type":"code","source":"show_img(cbb, 'Cassava Bacterial Blight (CBB)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:25:56.701928Z","iopub.execute_input":"2022-01-12T10:25:56.704592Z","iopub.status.idle":"2022-01-12T10:25:57.862917Z","shell.execute_reply.started":"2022-01-12T10:25:56.704542Z","shell.execute_reply":"2022-01-12T10:25:57.862051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_hist(cbb, 'Cassava Bacterial Blight (CBB)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:25:57.864414Z","iopub.execute_input":"2022-01-12T10:25:57.864855Z","iopub.status.idle":"2022-01-12T10:26:00.924021Z","shell.execute_reply.started":"2022-01-12T10:25:57.864815Z","shell.execute_reply":"2022-01-12T10:26:00.920959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_box_plot(cbb, 'Cassava Bacterial Blight (CBB)')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T10:26:00.925908Z","iopub.execute_input":"2022-01-12T10:26:00.926550Z","iopub.status.idle":"2022-01-12T10:26:15.310255Z","shell.execute_reply.started":"2022-01-12T10:26:00.926488Z","shell.execute_reply":"2022-01-12T10:26:15.309334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the following distribution of median values for pixel intensities:\n\n- Red: 102\n- Green: 117\n- Blue: 66","metadata":{}},{"cell_type":"markdown","source":"<h1 style='background:#2cab6c; border:0; color:white'><center>Conclusion</center></h1>","metadata":{}},{"cell_type":"markdown","source":"The dataset contains 21397 images. There are 5 class labels in total, including 4 disease labels and one plant health label. From EDA we see that the training labels are highly imbalanced: more than 60% of the training labels are in the CMD class, and only about 5% of them are in the CBB class. So far, we cannot predict how an imbalance in the dataset will affect the accuracy of the classification model.\n\nThe size of the images is 600 by 800 pixels. In training, we will reduce the size of the images to train the model more efficiently.\n\nAlso, by analyzing the histograms of color channels, we see that diseases are always accompanied by obvious discrepancies in color. Namely:\n\n- CGM class images have the highest median values of RGB channels;\n- CBB class images have the lowest median values of RGB channels.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}