{"cells":[{"metadata":{},"cell_type":"markdown","source":"To see what we have here I decided to just use code from my other notebooks. \nLet's see what we can find out."},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd \nimport imageio\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom PIL import Image, ImageOps\nimport scipy.ndimage as ndi","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#here we need to get to the data\ndirname = '/kaggle/input/herbarium-2020-fgvc7/nybg2020/train/images'\ndir_001 = os.path.join(dirname, '001')\ndir_001_19 = os.path.join(dir_001, '19')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"os.listdir(dir_001)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"os.listdir(dir_001_19)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"I'm going to have fun and try to play with this images. And try to get to know this dataset. "},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_imgs_one_dir(item_dir, num_imgs=25):\n    all_item_dirs = os.listdir(item_dir)\n    item_files = [os.path.join(item_dir, file) for file in all_item_dirs][:num_imgs]\n\n    plt.figure(figsize=(10, 10))\n    for idx, img_path in enumerate(item_files):\n        plt.subplot(5, 5, idx+1)\n\n        img = plt.imread(img_path)\n        plt.imshow(img)\n\n    plt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_imgs_one_dir(dir_001_19)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_img_hist_3channel(item_dir, num_img=6):\n    all_item_dirs = os.listdir(item_dir)\n    item_files = [os.path.join(item_dir, file) for file in all_item_dirs][:num_img]\n\n    #plt.figure(figsize=(10, 10))\n    for idx, img_path in enumerate(item_files):\n        fig1 = plt.figure(idx,figsize=(10, 10))\n        fig1.add_subplot(2, 2, 1)\n        img = mpimg.imread(img_path, )\n        plt.imshow(img)\n        fig1.add_subplot(2, 2, 2)\n        plt.hist(img.ravel(), bins = 256, color = 'orange')\n        plt.hist(img[:, :, 0].ravel(), bins = 256, color = 'red')\n        plt.hist(img[:, :, 1].ravel(), bins = 256, color = 'green')\n        plt.hist(img[:, :, 2].ravel(), bins = 256, color = 'blue')\n        plt.xlabel('Intensity')\n        plt.ylabel('Count')\n        plt.legend(['Total', 'Red Channel', 'Green Channel', 'Blue Channel'])\n        plt.show()\n    \n    plt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_img_hist_3channel(dir_001_19)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"I wonder what would happen if we equalize this histograms. Note that we only play with one image folder. "},{"metadata":{"trusted":true},"cell_type":"code","source":"os.mkdir('/kaggle/working/work_imgs')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dir_output = os.path.join('/kaggle/working/','work_imgs');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def  hist_equal(path_from, path_to, stop=4):\n    i=1\n    files = os.listdir(path_from)\n    \n    for file in files[:stop]: \n        try:\n            file_dir = os.path.join(path_from, file)\n            file_dir_save = os.path.join(path_to, file)\n            img = Image.open(file_dir)\n            img = ImageOps.equalize(img)\n            #img = img.convert(\"RGB\") #konwersja z RGBA do RGB, usuniecie kanału alfa zeby zapisać do jpg\n            img.save(file_dir_save) \n            i=i+1\n        except:\n            continue","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"hist_equal(dir_001_19, dir_output)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We check only first 4 of the images. "},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_img_hist_3channel(dir_output, 4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"No idea if this will be helpful in the future. But who cares. :D"},{"metadata":{},"cell_type":"markdown","source":"We can also see cumulative distribution function (CDF) and find out how it changed after equalisation."},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_cdf_comparison(item_dir_before,item_dir_after, num_img=1):\n    all_item_dirs = os.listdir(item_dir_before)\n    item_files_before = [os.path.join(item_dir_before, file) for file in all_item_dirs][:num_img]\n    item_files_after = [os.path.join(item_dir_after, file) for file in all_item_dirs][:num_img]\n  \n  #plt.figure(figsize=(10, 10))\n    for idx, img_path in enumerate(item_files_before):\n        im_b = imageio.imread(img_path)\n        hist_b = ndi.histogram(im_b, min=0, max=255, bins=256)\n        cdf_b = hist_b.cumsum() / hist_b.sum()\n        \n        img_path_a = item_files_after[idx]\n        im_a = imageio.imread(img_path_a)\n        hist_a = ndi.histogram(im_a, min=0, max=255, bins=256)\n        cdf_a = hist_a.cumsum() / hist_a.sum()\n\n        fig1 = plt.figure(idx,figsize=(10, 10))\n        fig1.add_subplot(2, 4, 1)\n        img_b = mpimg.imread(img_path, )\n        plt.title(\"Before. {}\".format(idx))\n        plt.imshow(img_b, cmap='gray')\n        fig1.add_subplot(2, 4, 4)\n        plt.title(\"CDF comparison\")\n        plt.plot(cdf_b)\n        \n        fig2 = plt.figure(idx,figsize=(10, 10))\n        fig2.add_subplot(2, 4, 2)\n        img_a = mpimg.imread(img_path_a, )\n        plt.title(\"Before. {}\".format(idx))\n        plt.imshow(img_a)\n        fig1.add_subplot(2, 4, 4)\n        plt.plot(cdf_a)\n        plt.legend(['CDF before', 'CDF after'])\n\n    plt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_cdf_comparison(dir_001_19, dir_output, 4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"To be continued... ;)"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}