{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport json\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib.image as mpimg\nfrom skimage import color\nimport os","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_images = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/train.csv')\nsample_sub = pd.read_csv('/kaggle/input/cassava-leaf-disease-classification/sample_submission.csv')\nTRAIN_IMAGES_PATH = '/kaggle/input/cassava-leaf-disease-classification/train_images'\nTEST_IMAGES_PATH = '../input/cassava-leaf-disease-classification/test_images'\n\nwith open('../input/cassava-leaf-disease-classification/label_num_to_disease_map.json') as json_data:\n    label_map = json.load(json_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(train_images['label'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This plot indicates that we have a class imbalance in the dataset."},{"metadata":{"trusted":true},"cell_type":"code","source":"label_map","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def select_imgs(n, label):\n    '''Function to select random ids from the dataframe given a particular label'''\n    t = train_images[train_images['label'] == label]\n    img_ids = t.sample(n = n, random_state = 0)['image_id']\n    return list(img_ids)\n\ndef plot_images(df, ids, label = None):\n    '''Plots an even number of images in 2 rows'''\n    n = len(ids)\n    fig, ax = plt.subplots(2, n//2, figsize = (20,10))\n    for i, im_id in enumerate(ids):\n        img = mpimg.imread(os.path.join(TRAIN_IMAGES_PATH, im_id))\n        ax[i//(n//2)][i%(n//2)].imshow(img)\n        ax[i//(n//2)][i%(n//2)].axis('off')\n    plt.tight_layout()\n    if label is not None:\n        plt.suptitle(label_map[str(label)])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#some label 0 images\nplot_images(train_images, select_imgs(8, 0), 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#some label 1 images\nplot_images(train_images, select_imgs(8, 1), 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#some label 2 images\nplot_images(train_images, select_imgs(8, 2), 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#some label 3 images\nplot_images(train_images, select_imgs(8, 3), 3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#some label 4 images\nplot_images(train_images, select_imgs(8, 4), 4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Looking at these images we realise that the colour of the diseased leaves is one of the key identifiers of the disease. Training networks with grayscale images may not perform well."},{"metadata":{},"cell_type":"markdown","source":"We shall now illustrate the differences in the brightness across all images."},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_brightness(image):\n    image = color.rgb2gray(image)\n    return np.mean(image)*255\n\n#get brightness of each image and append to dataframe\nbrightness_array = []\nimage_list = list(train_images['image_id'].unique())\nfor img in image_list:\n    image = mpimg.imread(os.path.join(TRAIN_IMAGES_PATH, img))\n    brightness = get_brightness(image)\n    brightness_array.append(brightness)\n\ndf = pd.DataFrame({'image_id': image_list,\n                         'brightness': brightness_array})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Mean Brightness is: ', df['brightness'].mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Max Brightness is: ', df['brightness'].max())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Min Brightness is: ', df['brightness'].min())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.hist(df['brightness'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#bright images\nbright_ids = df[df['brightness'] > 160].image_id\nplot_images(train_images, bright_ids[0:8])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#dark ids\ndark_ids = df[df['brightness'] < 50].image_id\nplot_images(train_images, dark_ids[0:8])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}