{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n#print(os.listdir(\"../input\"))\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os, sys\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport skimage.io\nfrom skimage.transform import resize\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport PIL\nfrom PIL import Image, ImageOps\nimport cv2\nfrom sklearn.utils import class_weight, shuffle\nfrom keras.losses import binary_crossentropy\nfrom keras.applications.resnet50 import preprocess_input\nimport keras.backend as K\nimport tensorflow as tf\nfrom sklearn.metrics import f1_score, fbeta_score\nfrom keras.utils import Sequence\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\n\nWORKERS = 2\nCHANNEL = 3\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nIMG_SIZE = 512\nNUM_CLASSES = 5\nSEED = 77\nTRAIN_NUM = 1000 # use 1000 when you just want to explore new idea, use -1 for full train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(train_df['diagnosis'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df0 = train_df[train_df['diagnosis']==0].sample(10,random_state=SEED)\ndf1 = train_df[train_df['diagnosis']==1].sample(10,random_state=SEED)\ndf2 = train_df[train_df['diagnosis']==2].sample(10,random_state=SEED)\ndf3 = train_df[train_df['diagnosis']==3].sample(10,random_state=SEED)\ndf4 = train_df[train_df['diagnosis']==4].sample(10,random_state=SEED)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#plt.imshow('../input/train_images/7b20210d9120.png')\nskimage.io.imshow('../input/train_images/7b20210d9120.png')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,20))\nfor count, i in enumerate(df0['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    skimage.io.imshow('../input/train_images/' + str(i) + '.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,20))\nfor count, i in enumerate(df1['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    skimage.io.imshow('../input/train_images/' + str(i) + '.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,20))\nfor count, i in enumerate(df2['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    skimage.io.imshow('../input/train_images/' + str(i) + '.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,20))\nfor count, i in enumerate(df3['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    skimage.io.imshow('../input/train_images/' + str(i) + '.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,20))\nfor count, i in enumerate(df4['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    skimage.io.imshow('../input/train_images/' + str(i) + '.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Check heatmap"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(30,15))\nfor count, i in enumerate(df4['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    img = skimage.io.imread('../input/train_images/' + str(i) + '.png')\n    gray = skimage.color.rgb2gray(img)\n    sns.heatmap(gray,cbar=False,xticklabels=False, yticklabels=False)\n    if count == 5:\n        break\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(30,15))\nfor count, i in enumerate(df0['id_code']):\n    ax1 = fig.add_subplot(2,5,count+1)\n    img = skimage.io.imread('../input/train_images/' + str(i) + '.png')\n    gray = skimage.color.rgb2gray(img)\n    sns.heatmap(gray,cbar=False,xticklabels=False, yticklabels=False)\n    if count == 5:\n        break\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## Histogram"},{"metadata":{"trusted":true},"cell_type":"code","source":"def imghistplotter(imglist, normflag):\n    fig = plt.figure(figsize=(30,20))\n    for count,img in enumerate(imglist):\n        ax = fig.add_subplot(2,len(imglist)/2,count+1)\n        img_b = img[:,:,0].reshape(img.shape[0]*img.shape[1])\n        img_g = img[:,:,1].reshape(img.shape[0]*img.shape[1])\n        img_r = img[:,:,2].reshape(img.shape[0]*img.shape[1])\n        plt.hist(img_r,bins=255,color='red',normed=normflag)\n        plt.hist(img_g,bins=255,color='green',normed=normflag)\n        plt.hist(img_b,bins=255,color='blue',normed=normflag)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imglist = [cv2.imread('../input/train_images/' + str(i) + '.png') for i in df0['id_code']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imghistplotter(imglist, normflag=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imglist_4 = [cv2.imread('../input/train_images/' + str(i) + '.png') for i in df4['id_code']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imghistplotter(imglist_4,normflag=True)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}