{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n       # print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## This notebook is a culmination of various techniques explored by Kagglers - [ilovescience](https://www.kaggle.com/tanlikesmath/cassava-classification-eda-fastai-starter), [Kevin Joseph Scaria](https://www.kaggle.com/kevinscaria/cassava-leaf-disease-eda-modelling), [Yaroslav Isaienkov](https://www.kaggle.com/ihelon/cassava-leaf-disease-exploratory-data-analysis), [Aayush Jain](https://www.kaggle.com/foolofatook/starter-eda-cassava-leaf-disease)"},{"metadata":{},"cell_type":"markdown","source":"# Loading Data & Basic Visualization\n\n## Importing Libraries"},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\nimport json\nimport seaborn as sns\nimport albumentations as A\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\n\nBASE_DIR = \"../input/cassava-leaf-disease-classification/\"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Let's have a look at the class labels in the dataset i.e Disease Names"},{"metadata":{"_kg_hide-output":false,"trusted":true},"cell_type":"code","source":"label_mapping = pd.read_json(os.path.join(BASE_DIR, 'label_num_to_disease_map.json'), typ = 'series')\nprint(label_mapping)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Reading the CSV file which has details on the images and lables"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(os.path.join(BASE_DIR, \"train.csv\"))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Creating a mapping from image_id and label -> Path of the image (Usefull for segregating the images label wise in different folders)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create a new column that holds the path of each image\ndf['path'] = df['image_id'].map(lambda x: os.path.join(BASE_DIR, 'train_imgs', x))\n\n# Shuffle the Dataframe\ndf = df.sample(frac=1).reset_index(drop=True) ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Mapping the label -> Class name"},{"metadata":{"trusted":true},"cell_type":"code","source":"df[\"class_name\"] = df[\"label\"].map(label_mapping)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Total number of images in the dataset - {}'.format(len(df)))\ndf","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualizing the distribution of data"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8, 4))\nsns.countplot(y = \"class_name\", data = df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Visualizations\n\n## Reading a single image and it's shape"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"img = cv2.imread(os.path.join(BASE_DIR, 'train_images', '3613193696.jpg'))\nplt.imshow(img)\nprint('Shape of the Image - {}'.format(img.shape))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Function for easy class-wise visualization"},{"metadata":{"trusted":true},"cell_type":"code","source":"def visualize_batch(image_ids, labels):\n    plt.figure(figsize=(16, 12))\n    \n    for ind, (image_id, label) in enumerate(zip(image_ids, labels)):\n        plt.subplot(3, 3, ind + 1)\n        image = cv2.imread(os.path.join(BASE_DIR, \"train_images\", image_id))\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        plt.imshow(image)\n        plt.title(f'Class: {label}', fontsize = 12)\n        plt.axis(\"off\")\n    \n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Random sampling"},{"metadata":{"trusted":true},"cell_type":"code","source":"random_df = df.sample(9)\nimage_ids = random_df[\"image_id\"].values\nlabels = random_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization of Images - Class : Cassava Bacterial Blight (CBB)"},{"metadata":{"trusted":true},"cell_type":"code","source":"cbb_df = df[df[\"label\"] == 0]\nprint('Total train images for Class {} - {}'.format(label_mapping[0], cbb_df.shape[0]))\n\ncbb_df = cbb_df.sample(9)\nimage_ids = cbb_df[\"image_id\"].values\nlabels = cbb_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization of Images - Class : Cassava Brown Steak Disease (CBSD)"},{"metadata":{"trusted":true},"cell_type":"code","source":"cbsd_df = df[df[\"label\"] == 1]\nprint('Total train images for Class {} - {}'.format(label_mapping[1], cbsd_df.shape[0]))\n\ncbsd_df = cbsd_df.sample(9)\nimage_ids = cbsd_df[\"image_id\"].values\nlabels = cbsd_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization of Images - Class : Cassava Green Mottle (CGM)"},{"metadata":{"trusted":true},"cell_type":"code","source":"cgm_df = df[df[\"label\"] == 2]\nprint('Total train images for Class {} - {}'.format(label_mapping[2], cgm_df.shape[0]))\n\ncgm_df = cgm_df.sample(9)\nimage_ids = cgm_df[\"image_id\"].values\nlabels = cgm_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization of Images - Class : Cassava Mosaic Disease (CMD)"},{"metadata":{"trusted":true},"cell_type":"code","source":"cmd_df = df[df[\"label\"] == 3]\nprint('Total train images for Class {} - {}'.format(label_mapping[3], cmd_df.shape[0]))\n\ncmd_df = cmd_df.sample(9)\nimage_ids = cmd_df[\"image_id\"].values\nlabels = cmd_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualization of Images - Class : Healthy (H)"},{"metadata":{"trusted":true},"cell_type":"code","source":"h_df = df[df[\"label\"] == 4]\nprint('Total train images for Class {} - {}'.format(label_mapping[4], h_df.shape[0]))\n\nh_df = h_df.sample(9)\nimage_ids = h_df[\"image_id\"].values\nlabels = h_df[\"class_name\"].values\n\nvisualize_batch(image_ids, labels)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploring Data Augmentations\n\n## Augmentations in TensorFlow"},{"metadata":{"trusted":true},"cell_type":"code","source":"def augmentImage(image_id, seed = 0):\n    \n    # Reading image given the path\n    image = tf.io.read_file(os.path.join(BASE_DIR, \"train_images\", image_id))\n    image = tf.image.decode_jpeg(image, channels = 3)\n    \n    original_image = image\n    \n    # Brightness, Contrast\n    brightness = tf.image.random_brightness(image, 0.4, seed = seed)\n    contrast = tf.image.random_contrast(image, 0.2, 0.6, seed = seed)\n    \n    # Cropping\n    crop = tf.image.random_crop(image, size = [500, 600, 3], seed = seed)\n    central_crop = tf.image.central_crop(image, central_fraction = 0.3)\n    \n    \n    # Flippings\n    # replace with random_flip_left_right when using as part of a augmentation pipeline\n    left_right = tf.image.flip_left_right(image)    \n    #replace with random_flip_up_down when using as part of a augmentation pipeline\n    up_down = tf.image.flip_up_down(image)\n    transpose = tf.image.transpose(image)\n    \n    # Color-based\n    hue = tf.image.random_hue(image, 0.2, seed = seed)\n    saturation = tf.image.random_saturation(image, 5, 10, seed = seed)\n    jpeg_quality = tf.image.random_jpeg_quality(image, 75, 85)\n        \n    # Rotations\n    #rotated = tf.image.rot90(image)    \n    \n    # Grayscale\n    grayscaled = tf.image.rgb_to_grayscale(image)\n    \n    \n    return (original_image, brightness, contrast, crop, central_crop, left_right, up_down, transpose, hue, saturation, jpeg_quality)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def displayAugmentations(l, index):\n    \n    augmentedImages = augmentImage(l[index])\n    plt.figure(figsize=(16, 12))\n    for i, imageName in zip(range(11), ['Input Image', 'Augmented - Brightness', 'Augmented - Contrast', 'Augmented - Crop', 'Augmented - Center Cropped', 'Augmented - Horizontal Flip',\n                                      'Augmented - Vertical Flip', 'Augmented - Transpose' , 'Augmented - Hue', 'Augmented - Saturation','Augmented - Jpeg Quality']):\n        ax = plt.subplot(3, 4, i + 1)\n        plt.imshow(augmentedImages[i].numpy().astype(\"uint8\"))\n        plt.title(imageName)\n        plt.axis(\"off\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cbb_list = cbb_df['image_id'].tolist()\ncbsd_list = cbsd_df['image_id'].tolist()\ncgm_list = cgm_df['image_id'].tolist()\ncmd_list = cmd_df['image_id'].tolist()\nh_list = h_df['image_id'].tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Bacterial Blight Images')\ndisplayAugmentations(cbb_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Brown Steak Disease Images')\ndisplayAugmentations(cbsd_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Green Mottle Images')\ndisplayAugmentations(cgm_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Mosaic Disease Images')\ndisplayAugmentations(cmd_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Healthy Leaf Images')\ndisplayAugmentations(h_list, 0)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Augmentations in Albumentations"},{"metadata":{"trusted":true},"cell_type":"code","source":"def apply_augmentation(image_id):\n    \n    img = cv2.imread(os.path.join(BASE_DIR, \"train_images\", image_id))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    coarse_dropout = A.CoarseDropout(p = 1.0, max_holes = 100, max_height = 50, max_width = 50, min_holes = 35, min_height = 10, min_width = 10)\n    clahe = A.CLAHE (clip_limit = 4.0, tile_grid_size=(8, 8), always_apply = False, p = 0.7)\n    c_crop = A.CenterCrop(600, 375, always_apply = False, p = 1.0)\n    gauss_noise = A.GaussNoise(var_limit=(10.0, 160.0), mean = 25, always_apply = False, p = 0.7)\n    distort = A.GridDistortion(always_apply = False, p = 1.0, num_steps = 6, distort_limit=(-0.40, 0.59), interpolation = 1, border_mode = 0, value = (0, 0, 0), mask_value = None)\n    h_flip = A.HorizontalFlip(always_apply = False, p = 1.0)\n    bright = A.RandomBrightness(always_apply = False, p = 1.0, limit=(-0.30, 0.43))\n    contrast = A.RandomContrast(always_apply = False, p = 1.0, limit=(-0.38, 0.30))\n    fog = A.RandomFog(always_apply = False, p = 1.0, fog_coef_lower=0.07, fog_coef_upper=0.26, alpha_coef = 1.0)\n    rain = A.RandomRain(always_apply = False, p = 1.0, slant_lower = 2, slant_upper = 4, drop_length = 7, drop_width = 1, drop_color = (0, 0, 0), blur_value = 1, brightness_coefficient = 0.97, rain_type = 'drizzle')\n    r_crop = A.RandomResizedCrop(always_apply = False, p = 1.0, height = 130, width = 229, scale = (0.08, 1.0), ratio = (0.75, 1.4), interpolation = 0)\n    rotate = A.RandomRotate90(always_apply = False, p = 1.0)\n    # To be tried - Cutout, Elastic Trnasform, ISONoise, HSV, MotionBlur , RandomSizedCrop, ShiftScaleRotate\n    \n    t_1 = coarse_dropout(image = img)[\"image\"]\n    t_2 = clahe(image = img)[\"image\"]\n    t_3 = c_crop(image = img)[\"image\"]\n    t_4 = gauss_noise(image = img)[\"image\"]\n    t_5 = distort(image = img)[\"image\"]\n    t_6 = h_flip(image = img)[\"image\"]\n    t_7 = bright(image = img)[\"image\"]\n    t_8 = contrast(image = img)[\"image\"]\n    t_9 = fog(image = img)[\"image\"]\n    t_10 = rain(image = img)[\"image\"]\n    t_11 = r_crop(image = img)[\"image\"]\n    t_12 = rotate(image = img)[\"image\"]\n        \n    return (img, t_1, t_2, t_3, t_4, t_5, t_6, t_7, t_8, t_9, t_10, t_11, t_12)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def displayAAugmentations(l, index):\n    \n    augmentedImages = apply_augmentation(l[index])\n    plt.figure(figsize=(16, 12))\n    for i, imageName in zip(range(13), ['Input Image', 'Augmented - Coarse Dropout', 'Augmented - CLAHE', 'Augmented - Center Crop', 'Augmented - Gaussian Noise', 'Augmented - Distortion',\n                                      'Augmented - Horizontal Flip', 'Augmented - Brightness' , 'Augmented - Contrast', 'Augmented - Fog', 'Augmented - Rain', 'Augmented - ResizedCrop', 'Augmented - Rotation']):\n        ax = plt.subplot(4, 4, i + 1)\n        plt.imshow(augmentedImages[i].astype(\"uint8\"))\n        plt.title(imageName)\n        plt.axis(\"off\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Bacterial Blight Images')\ndisplayAAugmentations(cbb_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Brown Steak Disease Images')\ndisplayAAugmentations(cbsd_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Green Mottle Images')\ndisplayAAugmentations(cgm_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Cassava Mosaic Disease Images')\ndisplayAAugmentations(cmd_list, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Healthy Leaf Images')\ndisplayAAugmentations(h_list, 0)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}