{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ntest = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\nsample_submission = pd.read_csv('../input/aptos2019-blindness-detection/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We have class imbalnce in our data set"},{"metadata":{"trusted":true},"cell_type":"code","source":"train['diagnosis'].value_counts().plot(kind='bar');\nplt.title('Class counts');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = 'No DR', 'Moderate', 'Mild', 'Proliferative DR', 'Severe'\nsizes = train.diagnosis.value_counts()\n\nfig1, ax1 = plt.subplots(figsize=(10,7))\nax1.pie(sizes, labels=labels, autopct='%1.1f%%', shadow=True, startangle=90)\nax1.axis('equal')\n\nplt.title('Diabetic retinopathy condition labels')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.columns\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_IMG_PATH = \"../input/aptos2019-blindness-detection/train_images\"\nTEST_IMG_PATH = \"../input/aptos2019-blindness-detection/test_images\"\n\n# function to plot a grid of images\n\ntemp = train[train['diagnosis'] == 1]\n\nimage_name = temp.iloc[0][0]\nfrom PIL import Image\nplt.figure(figsize=(14,6))\nplt.subplot(121)\nim = Image.open(TRAIN_IMG_PATH + '/' + image_name + '.png')\nplt.imshow(np.asarray(im))\nplt.title('No DR')\n\nplt.subplot(122)\ntemp = train[train['diagnosis'] == 4]\n\nimage_name = temp.iloc[0][0]\nim = Image.open(TRAIN_IMG_PATH + '/' + image_name + '.png')\nplt.imshow(np.asarray(im))\nplt.title('Severe DR')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_image_sizes(df, train = True):\n    '''\n    Function to get sizes of images from test and train sets.\n    INPUT:\n        df - dataframe containing image filenames\n        train - indicates whether we are getting sizes of images from train or test set\n    '''\n    if train:\n        path = TRAIN_IMG_PATH\n    else:\n        path = TEST_IMG_PATH\n        \n    widths = []\n    heights = []\n    \n    images = df.id_code\n    \n    max_im = Image.open(os.path.join(path, images[0] + '.png'))\n    min_im = Image.open(os.path.join(path, images[0] + '.png'))\n        \n    for im in range(0, len(images)):\n        image = Image.open(os.path.join(path, images[im] + '.png'))\n        width, height = image.size\n        \n        if len(widths) > 0:\n            if width > max(widths):\n                max_im = image\n\n            if width < min(widths):\n                min_im = image\n\n        widths.append(width)\n        heights.append(height)\n        \n    return widths, heights, max_im, min_im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_widths, train_heights, max_train, min_train = get_image_sizes(train, train = True)\ntest_widths, test_heights, max_test, min_test = get_image_sizes(test, train = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(14,6))\nplt.subplot(121)\nsns.distplot(train_widths, kde=False, label='Train Width')\nsns.distplot(train_heights, kde=False, label='Train Height')\nplt.legend()\nplt.title('Training Image Dimension Histogram', fontsize=15)\n\n\nplt.subplot(122)\nsns.distplot(test_widths, kde=False, label='Test Width')\nsns.distplot(test_heights, kde=False, label='Test Height')\nplt.legend()\nplt.title('Test Image Dimension Histogram', fontsize=15)\n\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}