{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport cv2 as cv\nfrom numpy.random import seed\nseed(45)\nimport pickle\n\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom skimage import io\nfrom glob import glob \n%matplotlib inline","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-11-11T19:42:58.371577Z","iopub.execute_input":"2021-11-11T19:42:58.371884Z","iopub.status.idle":"2021-11-11T19:43:06.273955Z","shell.execute_reply.started":"2021-11-11T19:42:58.371855Z","shell.execute_reply":"2021-11-11T19:43:06.27303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dirname = '/kaggle/input/histopathologic-cancer-detection/train'","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:06.276386Z","iopub.execute_input":"2021-11-11T19:43:06.276633Z","iopub.status.idle":"2021-11-11T19:43:06.280717Z","shell.execute_reply.started":"2021-11-11T19:43:06.276606Z","shell.execute_reply":"2021-11-11T19:43:06.27976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset exploration","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:06.282109Z","iopub.execute_input":"2021-11-11T19:43:06.282369Z","iopub.status.idle":"2021-11-11T19:43:06.90291Z","shell.execute_reply.started":"2021-11-11T19:43:06.282343Z","shell.execute_reply":"2021-11-11T19:43:06.902001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label distribution","metadata":{}},{"cell_type":"code","source":"train_labels['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:06.904564Z","iopub.execute_input":"2021-11-11T19:43:06.905174Z","iopub.status.idle":"2021-11-11T19:43:06.919693Z","shell.execute_reply.started":"2021-11-11T19:43:06.90513Z","shell.execute_reply":"2021-11-11T19:43:06.918545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display a DataFrame showing the proportion of observations with each \n# possible of the target variable (which is label). \n(train_labels.label.value_counts() / len(train_labels)).to_frame()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:06.924013Z","iopub.execute_input":"2021-11-11T19:43:06.924352Z","iopub.status.idle":"2021-11-11T19:43:07.061751Z","shell.execute_reply.started":"2021-11-11T19:43:06.924318Z","shell.execute_reply":"2021-11-11T19:43:07.060809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.info()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:07.06511Z","iopub.execute_input":"2021-11-11T19:43:07.065629Z","iopub.status.idle":"2021-11-11T19:43:07.105495Z","shell.execute_reply.started":"2021-11-11T19:43:07.065587Z","shell.execute_reply":"2021-11-11T19:43:07.10244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data is not entirely balanced, there is more negative samples than positive, by about 30 percent","metadata":{}},{"cell_type":"markdown","source":"# View Sample Images","metadata":{}},{"cell_type":"code","source":"positive_samples = train_labels.loc[train_labels['label'] == 1].sample(4)\nnegative_samples = train_labels.loc[train_labels['label'] == 0].sample(4)\npositive_images = []\nnegative_images = []\nfor sample in positive_samples['id']:\n    path = os.path.join(train_dirname, sample+'.tif')\n    img = cv.imread(path)\n    positive_images.append(img)\n        \nfor sample in negative_samples['id']:\n    path = os.path.join(train_dirname, sample+'.tif')\n    img = cv.imread(path)\n    negative_images.append(img)\n\nfig,axis = plt.subplots(2,4,figsize=(20,8))\nfig.suptitle('Dataset samples presentation plot',fontsize=20)\nfor i,img in enumerate(positive_images):\n    axis[0,i].imshow(img)\n    rect = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='g',facecolor='none', linestyle=':', capstyle='round')\n    axis[0,i].add_patch(rect)\naxis[0,0].set_ylabel('Positive samples', size='large')\nfor i,img in enumerate(negative_images):\n    axis[1,i].imshow(img)\n    rect = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='r',facecolor='none', linestyle=':', capstyle='round')\n    axis[1,i].add_patch(rect)\naxis[1,0].set_ylabel('Negative samples', size='large')\n    ","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:07.107277Z","iopub.execute_input":"2021-11-11T19:43:07.107767Z","iopub.status.idle":"2021-11-11T19:43:08.482906Z","shell.execute_reply.started":"2021-11-11T19:43:07.107724Z","shell.execute_reply":"2021-11-11T19:43:08.4818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA analysis for each channel","metadata":{}},{"cell_type":"markdown","source":"## Visulization of ditributions for AVERAGE pixel values","metadata":{}},{"cell_type":"code","source":"## Visulization of ditributions for AVERAGE pixel values\n\nimport matplotlib.image as mpimg\ndf = train_labels.sample(n=10000).reset_index()\ndf.head()\n\npositive_samples = df.loc[df['label'] == 1]\nnegative_samples = df.loc[df['label'] == 0]\n\n## channels pixels for positive samples\npos_r_pixels_mean = []\npos_g_pixels_mean = []\npos_b_pixels_mean = []\n\nfor i, row in positive_samples.iterrows():\n    \n    img = mpimg.imread(f'{train_dirname}/{row.id}.tif')  \n    r_pixels = round(img[:, :, 0].flatten().mean(),2)\n    g_pixels = round(img[:, :, 1].flatten().mean(),2)\n    b_pixels = round(img[:, :, 2].flatten().mean(),2)\n    \n    pos_r_pixels_mean.append(r_pixels)\n    pos_g_pixels_mean.append(g_pixels)\n    pos_b_pixels_mean.append(b_pixels)\n\n## channels pixels for negative samples\nneg_r_pixels_mean = []\nneg_g_pixels_mean = []\nneg_b_pixels_mean = []\n\nfor i, row in negative_samples.iterrows():\n    \n    img = mpimg.imread(f'{train_dirname}/{row.id}.tif')  \n    r_pixels = round(img[:, :, 0].flatten().mean(),2)\n    g_pixels = round(img[:, :, 1].flatten().mean(),2)\n    b_pixels = round(img[:, :, 2].flatten().mean(),2)\n    \n    neg_r_pixels_mean.append(r_pixels)\n    neg_g_pixels_mean.append(g_pixels)\n    neg_b_pixels_mean.append(b_pixels)\n    \nlen(neg_r_pixels_mean)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:43:08.484694Z","iopub.execute_input":"2021-11-11T19:43:08.485272Z","iopub.status.idle":"2021-11-11T19:44:13.925486Z","shell.execute_reply.started":"2021-11-11T19:43:08.485229Z","shell.execute_reply":"2021-11-11T19:44:13.924534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:44:13.927119Z","iopub.execute_input":"2021-11-11T19:44:13.927694Z","iopub.status.idle":"2021-11-11T19:44:13.936352Z","shell.execute_reply.started":"2021-11-11T19:44:13.927652Z","shell.execute_reply":"2021-11-11T19:44:13.935338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nr_of_bins = 256 #each possible pixel value will get a bin in the following histograms\nfig,axs = plt.subplots(3,2,sharey=True,figsize=(8,8),dpi=150)\nfig.suptitle('Histogram of average channel pixels', fontsize= 24)\n#RGB channels\naxs[0,0].hist(pos_r_pixels_mean,bins=nr_of_bins,density=True)\naxs[0,1].hist(neg_r_pixels_mean,bins=nr_of_bins,density=True)\naxs[1,0].hist(pos_g_pixels_mean,bins=nr_of_bins,density=True)\naxs[1,1].hist(neg_g_pixels_mean,bins=nr_of_bins,density=True)\naxs[2,0].hist(pos_b_pixels_mean,bins=nr_of_bins,density=True)\naxs[2,1].hist(neg_b_pixels_mean,bins=nr_of_bins,density=True)\n\n#Set image labels\naxs[0,0].set_title(\"Positive samples (N =\" + str(positive_samples.shape[0]) + \")\");\naxs[0,1].set_title(\"Negative samples (N =\" + str(negative_samples.shape[0]) + \")\");\naxs[0,1].set_ylabel(\"Red\",rotation='horizontal',labelpad=35,fontsize=12)\naxs[1,1].set_ylabel(\"Green\",rotation='horizontal',labelpad=35,fontsize=12)\naxs[2,1].set_ylabel(\"Blue\",rotation='horizontal',labelpad=35,fontsize=12)\nfor i in range(3):\n    axs[i,0].set_ylabel(\"Relative frequency\")\n","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:44:13.938093Z","iopub.execute_input":"2021-11-11T19:44:13.938433Z","iopub.status.idle":"2021-11-11T19:44:17.854023Z","shell.execute_reply.started":"2021-11-11T19:44:13.938403Z","shell.execute_reply":"2021-11-11T19:44:17.853298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visulization of ditributions for ALL pixel values","metadata":{}},{"cell_type":"code","source":"### I will sample a relatively smaller dataset for all pixel values\nimport matplotlib.image as mpimg\ndf = train_labels.sample(n=1000).reset_index()\ndf.head()\n\npositive_samples = df.loc[df['label'] == 1]\nnegative_samples = df.loc[df['label'] == 0]\n\n## channels pixels for positive samples\npos_r_pixels_all = []\npos_g_pixels_all = []\npos_b_pixels_all = []\n\nfor i, row in positive_samples.iterrows():\n    \n    img = mpimg.imread(f'{train_dirname}/{row.id}.tif')  \n    r_pixels = img[:, :, 0].flatten()\n    g_pixels = img[:, :, 1].flatten()\n    b_pixels = img[:, :, 2].flatten()\n    \n    pos_r_pixels_all.append(r_pixels)\n    pos_g_pixels_all.append(g_pixels)\n    pos_b_pixels_all.append(b_pixels)\n\n## channels pixels for negative samples\nneg_r_pixels_all = []\nneg_g_pixels_all = []\nneg_b_pixels_all = []\n\nfor i, row in negative_samples.iterrows():\n    \n    img = mpimg.imread(f'{train_dirname}/{row.id}.tif')  \n    r_pixels = img[:, :, 0].flatten()\n    g_pixels = img[:, :, 1].flatten()\n    b_pixels = img[:, :, 2].flatten()\n    \n    neg_r_pixels_all.append(r_pixels)\n    neg_g_pixels_all.append(g_pixels)\n    neg_b_pixels_all.append(b_pixels)\nlen(neg_r_pixels_all)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:44:17.855042Z","iopub.execute_input":"2021-11-11T19:44:17.855371Z","iopub.status.idle":"2021-11-11T19:44:24.281485Z","shell.execute_reply.started":"2021-11-11T19:44:17.855345Z","shell.execute_reply":"2021-11-11T19:44:24.28058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nr_of_bins = 256 #each possible pixel value will get a bin in the following histograms\nfig,axs = plt.subplots(3,2,sharey=True,figsize=(8,8))\nfig.suptitle('Histogram of actual channel pixels', fontsize= 24)\n#RGB channels\n#axs[0,0].hist(pos_r_pixels_all,bins=nr_of_bins,density=True)\n#axs[0,1].hist(neg_r_pixels_all,bins=nr_of_bins,density=True)\n#axs[1,0].hist(pos_g_pixels_all,bins=nr_of_bins,density=True)\n#axs[1,1].hist(neg_g_pixels_all,bins=nr_of_bins,density=True)\n#axs[2,0].hist(pos_b_pixels_all,bins=nr_of_bins,density=True)\n#axs[2,1].hist(neg_b_pixels_all,bins=nr_of_bins,density=True)\naxs[0,0].hist(pos_r_pixels_all,bins=nr_of_bins)\naxs[0,1].hist(neg_r_pixels_all,bins=nr_of_bins)\naxs[1,0].hist(pos_g_pixels_all,bins=nr_of_bins)\naxs[1,1].hist(neg_g_pixels_all,bins=nr_of_bins)\naxs[2,0].hist(pos_b_pixels_all,bins=nr_of_bins)\naxs[2,1].hist(neg_b_pixels_all,bins=nr_of_bins)\ncustom_xlim = (0, 256)\ncustom_ylim = (0, 1000)\n# Setting the values for all axes.\nplt.setp(axs, xlim=custom_xlim, ylim=custom_ylim)\n\n#Set image labels\naxs[0,0].set_title(\"Positive samples (N =\" + str(positive_samples.shape[0]) + \")\");\naxs[0,1].set_title(\"Negative samples (N =\" + str(negative_samples.shape[0]) + \")\");\naxs[0,1].set_ylabel(\"Red\",rotation='horizontal',labelpad=35,fontsize=12)\naxs[1,1].set_ylabel(\"Green\",rotation='horizontal',labelpad=35,fontsize=12)\naxs[2,1].set_ylabel(\"Blue\",rotation='horizontal',labelpad=35,fontsize=12)\nfor i in range(3):\n    axs[i,0].set_ylabel(\"Pixels Counts\")","metadata":{"execution":{"iopub.status.busy":"2021-11-11T19:44:24.284931Z","iopub.execute_input":"2021-11-11T19:44:24.285243Z","iopub.status.idle":"2021-11-11T20:09:32.933202Z","shell.execute_reply.started":"2021-11-11T19:44:24.285212Z","shell.execute_reply":"2021-11-11T20:09:32.932275Z"},"trusted":true},"execution_count":null,"outputs":[]}]}