{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline\n\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\n\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('../input/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_dir = '../input/'\ntrain_dir = os.path.join(base_dir, 'train_images/')\ntrain = pd.read_csv(os.path.join(base_dir, 'train.csv'))\ntrain['path'] = train['id_code'].map(lambda x: os.path.join(train_dir, '{}.png'.format(x)))\ntrain = train.drop(columns = ['id_code'])\ntrain = train.sample(frac=1).reset_index(drop=True) #shuffle dataframe\n\ntest_dir = os.path.join(base_dir, 'test_images/')\ntest = pd.read_csv(os.path.join(base_dir, 'test.csv'))\ntest['path'] = test['id_code'].map(lambda x: os.path.join(test_dir, '{}.png'.format(x)))\ntest = test.drop(columns = ['id_code'])\ntest = test.sample(frac=1).reset_index(drop=True) #shuffle dataframe\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"size of train set: {}\".format(len(train)))\nprint(\"size of test set: {}\".format(len(test)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['diagnosis'].hist(figsize=(10,5), bins=10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"we can see that a lot of train images are in class 0 "},{"metadata":{"trusted":true},"cell_type":"code","source":"train_width, train_height = [], []\nfor i in range(len(train)):\n    image = Image.open(train['path'][i])\n    width, height = image.size\n    train_width.append(width)\n    train_height.append(height)\n    \ntest_width, test_height = [], []\nfor i in range(len(test)):\n    image = Image.open(test['path'][i])\n    width, height = image.size\n    test_width.append(width)\n    test_height.append(height)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['width'] = train_width\ntrain['height'] = train_height\ntest['width'] = test_width\ntest['height'] = test_height","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for df in [train, test]:\n    df['width_height_ratio'] = df['height'] / df['width']\n    df['width_height_added'] = df['height'] + df['width']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(12,20))\nax1 = fig.add_subplot(3, 2, 1)\nax2 = fig.add_subplot(3, 2, 2)\nax3 = fig.add_subplot(3, 2, 3)\nax4 = fig.add_subplot(3, 2, 4)\nax5 = fig.add_subplot(3, 2, 5)\nax6 = fig.add_subplot(3, 2, 6)\n\nax1.hist(train['width'])\nax1.set_title(\"train width\")\n\nax3.hist(test['width'])\nax3.set_title(\"test width\")\n\nax2.hist(train['height'])\nax2.set_title(\"train height\")\n\nax4.hist(test['height'])\nax4.set_title(\"test height\")\n\nax5.hist(train['width_height_ratio'])\nax5.set_title(\"train width height ratio\")\n\nax6.hist(test['width_height_ratio'])\nax6.set_title(\"test width height ratio\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"+ comparison of train image's width, height and test image's width, height\n+ we can see the difference in the distribution"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(12,8))\nax1 = fig.add_subplot(2, 2, 1)\nax2 = fig.add_subplot(2, 2, 2)\n\nax1.hist(train['width_height_added'], bins=5)\nax1.set_title(\"train width height added\")\n\nax2.hist(test['width_height_added'], bins=5)\nax2.set_title(\"test width height added\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"+ a lot images in the test set are very small compared to train set"},{"metadata":{"trusted":true},"cell_type":"code","source":"train['width_height_added'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['width_height_added'].describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"you can compare it with simple statistics"},{"metadata":{"trusted":true},"cell_type":"code","source":"train.groupby(['diagnosis'])['width_height_added'].mean()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"you can see that class 0 in train set have relatively small size of width and height"},{"metadata":{"trusted":true},"cell_type":"markdown","source":"    # Visualinze Train Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_class0 = [] \ntrain_class1 = []\ntrain_class2 = []\ntrain_class3 = []\ntrain_class4 = []\n\nnum_sample = 10\n\nfor path in train[train.diagnosis == 0].sample(num_sample)['path']:\n    im = Image.open(path)\n    train_class0.append(im)\n    \nfor path in train[train.diagnosis == 1].sample(num_sample)['path']:\n    im = Image.open(path)\n    train_class1.append(im)\n    \nfor path in train[train.diagnosis == 2].sample(num_sample)['path']:\n    im = Image.open(path)\n    train_class2.append(im)\n    \nfor path in train[train.diagnosis == 3].sample(num_sample)['path']:\n    im = Image.open(path)\n    train_class3.append(im)\n    \nfor path in train[train.diagnosis == 4].sample(num_sample)['path']:\n    im = Image.open(path)\n    train_class4.append(im)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train class 0 (No DR)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\ncolumns = 5\n\nfor i, image in enumerate(train_class0):\n    plt.subplot(len(train_class0) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train class 1 (Mild)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\ncolumns = 5\n\nfor i, image in enumerate(train_class1):\n    plt.subplot(len(train_class1) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train class 2 (Moderate)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\ncolumns = 5\n\nfor i, image in enumerate(train_class2):\n    plt.subplot(len(train_class2) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train class 3 (Severe)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\ncolumns = 5\n\nfor i, image in enumerate(train_class3):\n    plt.subplot(len(train_class3) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train class 4 (Proliferative DR)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\ncolumns = 5\n\nfor i, image in enumerate(train_class4):\n    plt.subplot(len(train_class4) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Visualize Test Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# width + height < 1500 => small image\n# width + height > 4000 => large image\n\ntest_image_small = []\ntest_image_large = []\n\nfor path in test[test.width_height_added < 1500].sample(30)['path']:\n    im = Image.open(path)\n    test_image_small.append(im)\n    \nfor path in test[test.width_height_added > 4000].sample(30)['path']:\n    im = Image.open(path)\n    test_image_large.append(im)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## test images (small size)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,20))\ncolumns = 5\n\nfor i, image in enumerate(test_image_small):\n    plt.subplot(len(test_image_small) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"+ small size of test images have class of near to 4 (seeing with my eyes)\n+ in the training set, class 0 were likely to be small image size. meaning **class distribution is very different in train and test set**"},{"metadata":{},"cell_type":"markdown","source":"## test images (large size)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,20))\ncolumns = 5\n\nfor i, image in enumerate(test_image_large):\n    plt.subplot(len(test_image_large) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"lets see the same thing for the train images"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_image_small = []\ntrain_image_large = []\n\nfor path in train[train.width_height_added < 1500].sample(30)['path']:\n    im = Image.open(path)\n    train_image_small.append(im)\n    \nfor path in train[train.width_height_added > 4000].sample(30)['path']:\n    im = Image.open(path)\n    train_image_large.append(im)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train images (small size)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,20))\ncolumns = 5\n\nfor i, image in enumerate(train_image_small):\n    plt.subplot(len(train_image_small) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## train images (large size)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,20))\ncolumns = 5\n\nfor i, image in enumerate(train_image_large):\n    plt.subplot(len(train_image_large) / columns + 1, columns, i + 1)\n    plt.imshow(image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Conclusion\n\n+ by looking into the dataset, we can clearly see the difference between the train and test dataset\n+ hopefully this simple eda explains the large gap between the cv score and lb score\n+ and it will be your job to make stable cross validation"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}