{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport shutil\n\nfrom glob import glob \nfrom skimage.io import imread\nimport gc\n\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils import to_categorical","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:08:44.450648Z","iopub.execute_input":"2021-07-17T13:08:44.451015Z","iopub.status.idle":"2021-07-17T13:08:44.458761Z","shell.execute_reply.started":"2021-07-17T13:08:44.45097Z","shell.execute_reply":"2021-07-17T13:08:44.457693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(os.listdir(\"../input/histopathologic-cancer-detection/train/\")))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:08:47.613825Z","iopub.execute_input":"2021-07-17T13:08:47.614162Z","iopub.status.idle":"2021-07-17T13:08:51.402576Z","shell.execute_reply.started":"2021-07-17T13:08:47.614127Z","shell.execute_reply":"2021-07-17T13:08:51.401614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_tile_dir = '../input/histopathologic-cancer-detection/train/'\ndf=pd.DataFrame({'path': glob(os.path.join(base_tile_dir,'*.tif'))})\n\ndf['id'] = df.path.map(lambda x: x.split('/')[4].split(\".\")[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:08:51.404255Z","iopub.execute_input":"2021-07-17T13:08:51.40465Z","iopub.status.idle":"2021-07-17T13:08:52.464868Z","shell.execute_reply.started":"2021-07-17T13:08:51.404608Z","shell.execute_reply":"2021-07-17T13:08:52.464052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:08:55.373725Z","iopub.execute_input":"2021-07-17T13:08:55.37405Z","iopub.status.idle":"2021-07-17T13:08:55.397425Z","shell.execute_reply.started":"2021-07-17T13:08:55.374018Z","shell.execute_reply":"2021-07-17T13:08:55.396645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\ndf_data = df.merge(labels, on = \"id\")","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:08:58.279343Z","iopub.execute_input":"2021-07-17T13:08:58.279673Z","iopub.status.idle":"2021-07-17T13:08:59.060305Z","shell.execute_reply.started":"2021-07-17T13:08:58.279642Z","shell.execute_reply":"2021-07-17T13:08:59.059417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:09:01.478872Z","iopub.execute_input":"2021-07-17T13:09:01.479215Z","iopub.status.idle":"2021-07-17T13:09:01.495138Z","shell.execute_reply.started":"2021-07-17T13:09:01.479186Z","shell.execute_reply":"2021-07-17T13:09:01.494191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing this image because it caused a training error previously\ndf_data = df_data[df_data['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']\n\n# removing this image because it's black\ndf_data = df_data[df_data['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']\ndf_data","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:09:04.25848Z","iopub.execute_input":"2021-07-17T13:09:04.258798Z","iopub.status.idle":"2021-07-17T13:09:04.383987Z","shell.execute_reply.started":"2021-07-17T13:09:04.258766Z","shell.execute_reply":"2021-07-17T13:09:04.383055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split X and y in train/test and build folders**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:09:59.293986Z","iopub.execute_input":"2021-07-17T13:09:59.294343Z","iopub.status.idle":"2021-07-17T13:09:59.419371Z","shell.execute_reply.started":"2021-07-17T13:09:59.294311Z","shell.execute_reply":"2021-07-17T13:09:59.418562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:01.818612Z","iopub.execute_input":"2021-07-17T13:10:01.818943Z","iopub.status.idle":"2021-07-17T13:10:01.866639Z","shell.execute_reply.started":"2021-07-17T13:10:01.818902Z","shell.execute_reply":"2021-07-17T13:10:01.865409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.groupby('label').count()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:06.258794Z","iopub.execute_input":"2021-07-17T13:10:06.259168Z","iopub.status.idle":"2021-07-17T13:10:06.323402Z","shell.execute_reply.started":"2021-07-17T13:10:06.259134Z","shell.execute_reply":"2021-07-17T13:10:06.32242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"label\",data=df_data)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:08.584522Z","iopub.execute_input":"2021-07-17T13:10:08.58485Z","iopub.status.idle":"2021-07-17T13:10:08.736032Z","shell.execute_reply.started":"2021-07-17T13:10:08.584819Z","shell.execute_reply":"2021-07-17T13:10:08.735039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_SIZE = 80000\n\n# take a random sample of class 0 with size equal to num samples in class 1\ndf_0 = df_data[df_data['label'] == 0].sample(SAMPLE_SIZE, random_state = 101)\ndf_0 = df_0.reset_index(drop=True)\n\n# filter out class 1\ndf_1 = df_data[df_data['label'] == 1].sample(SAMPLE_SIZE, random_state = 101)\ndf_1 = df_1.reset_index(drop=True)\n\n# concat the dataframes\ndf_data = shuffle(pd.concat([df_0, df_1], axis=0).reset_index(drop=True))\ndf_data = df_data.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:11.628734Z","iopub.execute_input":"2021-07-17T13:10:11.629082Z","iopub.status.idle":"2021-07-17T13:10:11.949738Z","shell.execute_reply.started":"2021-07-17T13:10:11.629031Z","shell.execute_reply":"2021-07-17T13:10:11.948831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_1.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:14.598691Z","iopub.execute_input":"2021-07-17T13:10:14.599173Z","iopub.status.idle":"2021-07-17T13:10:14.608513Z","shell.execute_reply.started":"2021-07-17T13:10:14.599132Z","shell.execute_reply":"2021-07-17T13:10:14.607735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_0.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:17.953858Z","iopub.execute_input":"2021-07-17T13:10:17.954221Z","iopub.status.idle":"2021-07-17T13:10:17.964348Z","shell.execute_reply.started":"2021-07-17T13:10:17.954189Z","shell.execute_reply":"2021-07-17T13:10:17.963406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:20.484473Z","iopub.execute_input":"2021-07-17T13:10:20.484811Z","iopub.status.idle":"2021-07-17T13:10:20.494497Z","shell.execute_reply.started":"2021-07-17T13:10:20.484781Z","shell.execute_reply":"2021-07-17T13:10:20.49361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = imread(df_1['path'][0], plugin='matplotlib')\nplt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:29:28.110261Z","iopub.execute_input":"2021-07-17T13:29:28.110657Z","iopub.status.idle":"2021-07-17T13:29:28.272375Z","shell.execute_reply.started":"2021-07-17T13:29:28.110624Z","shell.execute_reply":"2021-07-17T13:29:28.27149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"label\",data=df_data)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:26.718738Z","iopub.execute_input":"2021-07-17T13:10:26.719132Z","iopub.status.idle":"2021-07-17T13:10:26.861089Z","shell.execute_reply.started":"2021-07-17T13:10:26.719077Z","shell.execute_reply":"2021-07-17T13:10:26.860161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**===============================**","metadata":{}},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:31.014006Z","iopub.execute_input":"2021-07-17T13:10:31.01443Z","iopub.status.idle":"2021-07-17T13:10:31.159954Z","shell.execute_reply.started":"2021-07-17T13:10:31.014391Z","shell.execute_reply":"2021-07-17T13:10:31.159173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readImage(path):\n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    return rgb_img","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:33.209298Z","iopub.execute_input":"2021-07-17T13:10:33.209645Z","iopub.status.idle":"2021-07-17T13:10:33.214405Z","shell.execute_reply.started":"2021-07-17T13:10:33.209612Z","shell.execute_reply":"2021-07-17T13:10:33.213482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cropImg(path,w=32,h=32):\n    image=readImage(path)\n    # center = image.shape / 2\n    x = 48 - w/2\n    y = 48 - h/2\n    return image[int(y):int(y+h), int(x):int(x+w)]/255","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:35.458762Z","iopub.execute_input":"2021-07-17T13:10:35.459092Z","iopub.status.idle":"2021-07-17T13:10:35.463621Z","shell.execute_reply.started":"2021-07-17T13:10:35.459059Z","shell.execute_reply":"2021-07-17T13:10:35.462812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cropImg(df_1['path'][1000]).shape\nplt.imshow(cropImg(df_1['path'][1000]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:38.093993Z","iopub.execute_input":"2021-07-17T13:10:38.094367Z","iopub.status.idle":"2021-07-17T13:10:38.267989Z","shell.execute_reply.started":"2021-07-17T13:10:38.094337Z","shell.execute_reply":"2021-07-17T13:10:38.267052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv2.imread(df_1['path'][1000]).shape","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:40.650621Z","iopub.execute_input":"2021-07-17T13:10:40.650952Z","iopub.status.idle":"2021-07-17T13:10:40.659825Z","shell.execute_reply.started":"2021-07-17T13:10:40.650919Z","shell.execute_reply":"2021-07-17T13:10:40.659023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:42.974579Z","iopub.execute_input":"2021-07-17T13:10:42.974936Z","iopub.status.idle":"2021-07-17T13:10:42.97916Z","shell.execute_reply.started":"2021-07-17T13:10:42.974903Z","shell.execute_reply":"2021-07-17T13:10:42.978196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(N,df):\n    \"\"\"\n    loads in first N rows of the dataframe df\n    returns: X: np array of  N images, where  X.shape = (N, 96,96,3) \n             y:  np array of N labels, where y.shape = (N, )\n    \"\"\"\n    y = np.array(df['label'])[:N]\n    X = np.zeros([N,32,32,3])\n#     X = np.zeros([N,96,96,3])\n#     in tqdm_notebook(df.iterrows(), total=N)\n\n    for i in tqdm(range(N)):\n        X[i] =  cropImg(df['path'][i])\n    return X, y","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:47.723618Z","iopub.execute_input":"2021-07-17T13:10:47.723948Z","iopub.status.idle":"2021-07-17T13:10:47.730553Z","shell.execute_reply.started":"2021-07-17T13:10:47.723919Z","shell.execute_reply":"2021-07-17T13:10:47.729699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:51.343524Z","iopub.execute_input":"2021-07-17T13:10:51.34384Z","iopub.status.idle":"2021-07-17T13:10:51.349654Z","shell.execute_reply.started":"2021-07-17T13:10:51.343809Z","shell.execute_reply":"2021-07-17T13:10:51.348645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X,y = load_data(160000, df_data)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:10:53.238395Z","iopub.execute_input":"2021-07-17T13:10:53.238713Z","iopub.status.idle":"2021-07-17T13:11:15.134098Z","shell.execute_reply.started":"2021-07-17T13:10:53.238683Z","shell.execute_reply":"2021-07-17T13:11:15.131588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'The shape of X is {X.shape}')\nprint(f\"The shape of y is {y.shape}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:25:46.801609Z","iopub.execute_input":"2021-07-10T05:25:46.802017Z","iopub.status.idle":"2021-07-10T05:25:46.808809Z","shell.execute_reply.started":"2021-07-10T05:25:46.801982Z","shell.execute_reply":"2021-07-10T05:25:46.807354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_data(mat, target, train_ratio):\n    # Get the number of rows in the training data\n    train_rows = int(len(mat) * train_ratio)\n    print('got the number of rows to train')\n    # Place the first `train_rows` shuffled rows into the training data \n    # and the remaining rows into the test data\n    rng = np.random.default_rng(1)\n    \n    shuffled_indices = np.arange(len(mat))\n    print('got shuffled indices')\n    # print(shuffled_indices)\n    rng.shuffle(shuffled_indices)\n    # print(type(shuffled_indices[:train_rows]))\n    print('shuffled shuffled_indices')\n    X_train = np.array([mat[i] for i in shuffled_indices[:train_rows]])\n    print('got Xtrain')\n    X_test  = np.array([mat[i] for i in shuffled_indices[train_rows:]])\n    print('got Xtest')\n    Y_train = target[shuffled_indices[:train_rows]]\n    print('got ytrain')\n    Y_test  = target[shuffled_indices[train_rows:]]\n    print('got ytest')\n    return X_train, X_test, Y_train, Y_test","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:17:53.747655Z","iopub.execute_input":"2021-07-10T05:17:53.748075Z","iopub.status.idle":"2021-07-10T05:17:53.756024Z","shell.execute_reply.started":"2021-07-10T05:17:53.748041Z","shell.execute_reply":"2021-07-10T05:17:53.755134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xtrain,Xtest, ytrain, ytest = split_data(X,y, 0.8)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:25:53.221688Z","iopub.execute_input":"2021-07-10T05:25:53.222111Z","iopub.status.idle":"2021-07-10T05:25:55.444126Z","shell.execute_reply.started":"2021-07-10T05:25:53.222077Z","shell.execute_reply":"2021-07-10T05:25:55.442923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, BatchNormalization, Activation\nfrom keras.layers import Conv2D, MaxPool2D","metadata":{"execution":{"iopub.status.busy":"2021-07-10T04:39:52.190455Z","iopub.execute_input":"2021-07-10T04:39:52.190884Z","iopub.status.idle":"2021-07-10T04:39:52.196096Z","shell.execute_reply.started":"2021-07-10T04:39:52.190848Z","shell.execute_reply":"2021-07-10T04:39:52.195085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kernel_size = (3,3)\npool_size= (2,2)\nfirst_filter = 32\nsecond_filter = 64\nthird_filter = 128\n\ndropout_conv = 0.3\ndropout_dense = 0.5\nclass AlexNet(Sequential):\n    def __init__(self, input_shape):\n        super().__init__()\n        #layer 1\n        self.add(Conv2D(first_filter, kernel_size, input_shape = input_shape))\n        self.add(BatchNormalization())\n        self.add(MaxPool2D(pool_size=pool_size))\n        self.add(Dropout(dropout_conv))\n\n        #layer 2\n        self.add(Conv2D(second_filter, kernel_size=kernel_size, use_bias = False ))\n        self.add(BatchNormalization())\n        self.add(Activation(\"relu\"))\n        self.add(MaxPool2D(pool_size = pool_size)) \n        self.add(Dropout(dropout_conv))\n        \n        #layer 3\n        self.add(Conv2D(third_filter, kernel_size, use_bias=False))\n        self.add(BatchNormalization())\n        self.add(Activation(\"relu\"))\n        self.add(MaxPool2D(pool_size = pool_size)) \n        self.add(Dropout(dropout_conv))\n\n        #end layer\n        self.add(Flatten())\n        self.add(Dense(256, use_bias=False))\n        self.add(BatchNormalization())\n        self.add(Activation(\"relu\"))\n        self.add(Dropout(dropout_dense))\n        self.add(Dense(1, activation = \"sigmoid\"))\n       \n    \n        self.compile(loss=keras.losses.binary_crossentropy,\n              optimizer=keras.optimizers.Adam(0.001), \n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:26:10.920786Z","iopub.execute_input":"2021-07-10T05:26:10.921214Z","iopub.status.idle":"2021-07-10T05:26:10.937647Z","shell.execute_reply.started":"2021-07-10T05:26:10.921178Z","shell.execute_reply":"2021-07-10T05:26:10.936148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AlexNet((32, 32, 3))\n# model = AlexNet((96, 96, 3))\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:26:14.930849Z","iopub.execute_input":"2021-07-10T05:26:14.931301Z","iopub.status.idle":"2021-07-10T05:26:15.096767Z","shell.execute_reply.started":"2021-07-10T05:26:14.931259Z","shell.execute_reply":"2021-07-10T05:26:15.0947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training the model\nmodel.fit(\n    Xtrain,\n    ytrain,\n    batch_size=50,\n    epochs=10,\n    validation_data=(Xtest, ytest), verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:26:28.19135Z","iopub.execute_input":"2021-07-10T05:26:28.191863Z","iopub.status.idle":"2021-07-10T05:50:37.985688Z","shell.execute_reply.started":"2021-07-10T05:26:28.191822Z","shell.execute_reply":"2021-07-10T05:50:37.98436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model_2_alexnet')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:51:24.482033Z","iopub.execute_input":"2021-07-10T05:51:24.482576Z","iopub.status.idle":"2021-07-10T05:51:27.772317Z","shell.execute_reply.started":"2021-07-10T05:51:24.482513Z","shell.execute_reply":"2021-07-10T05:51:27.771256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.patches as patches","metadata":{"execution":{"iopub.status.busy":"2021-07-10T04:57:15.448123Z","iopub.execute_input":"2021-07-10T04:57:15.448602Z","iopub.status.idle":"2021-07-10T04:57:15.45307Z","shell.execute_reply.started":"2021-07-10T04:57:15.448562Z","shell.execute_reply":"2021-07-10T04:57:15.452194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1=readImage(df_0['path'][0])","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:03:22.041752Z","iopub.execute_input":"2021-07-10T05:03:22.042224Z","iopub.status.idle":"2021-07-10T05:03:22.05217Z","shell.execute_reply.started":"2021-07-10T05:03:22.04218Z","shell.execute_reply":"2021-07-10T05:03:22.051077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int((test1.shape[0])/2)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:07:07.458945Z","iopub.execute_input":"2021-07-10T05:07:07.45942Z","iopub.status.idle":"2021-07-10T05:07:07.465653Z","shell.execute_reply.started":"2021-07-10T05:07:07.459365Z","shell.execute_reply":"2021-07-10T05:07:07.464816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(10,4))\n# fig, ax = plt.subplots(2,5, figsize=(20,8))\n\nax[0].imshow(readImage(df_0['path'][0]))\n# Create a Rectangle patch\nbox = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',facecolor='none', linestyle=':', capstyle='round')\nax[0].add_patch(box)\nax[0].set_ylabel('Negative samples', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T04:57:45.667355Z","iopub.execute_input":"2021-07-10T04:57:45.667732Z","iopub.status.idle":"2021-07-10T04:57:45.995368Z","shell.execute_reply.started":"2021-07-10T04:57:45.667704Z","shell.execute_reply":"2021-07-10T04:57:45.994324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(10,4))\n\nax[0].imshow(readImage(df_0['path'][1000]))\nbox = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',facecolor='none', linestyle=':', capstyle='round')\nax[0].add_patch(box)\nax[0].set_ylabel('Negative samples', size='large')\n\nax[1].imshow(cropImg(df_0['path'][1000]))\nax[1].set_ylabel('Negative samples Croped', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:16:01.098123Z","iopub.execute_input":"2021-07-10T05:16:01.098585Z","iopub.status.idle":"2021-07-10T05:16:01.467144Z","shell.execute_reply.started":"2021-07-10T05:16:01.098548Z","shell.execute_reply":"2021-07-10T05:16:01.465979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(10,4))\n\nax[0].imshow(readImage(df_1['path'][1000]))\nbox = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',facecolor='none', linestyle=':', capstyle='round')\nax[0].add_patch(box)\nax[0].set_ylabel('Positive samples', size='large')\n\nax[1].imshow(cropImg(df_1['path'][1000]))\nax[1].set_ylabel('Positive samples Croped', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T05:13:32.413364Z","iopub.execute_input":"2021-07-10T05:13:32.413819Z","iopub.status.idle":"2021-07-10T05:13:32.81651Z","shell.execute_reply.started":"2021-07-10T05:13:32.413783Z","shell.execute_reply":"2021-07-10T05:13:32.815465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**===============================**","metadata":{}},{"cell_type":"code","source":"# train_test_split # stratify=y creates a balanced validation set.\ny = df_data['label']\ndf_train, df_val = train_test_split(df_data, test_size=0.10, random_state=101, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T18:12:04.804178Z","iopub.execute_input":"2021-07-09T18:12:04.804627Z","iopub.status.idle":"2021-07-09T18:12:04.945057Z","shell.execute_reply.started":"2021-07-09T18:12:04.804589Z","shell.execute_reply":"2021-07-09T18:12:04.94402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Training Size:\",df_train.shape[0])\nprint(\"Testing Size:\",df_val.shape[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-09T18:56:55.751585Z","iopub.execute_input":"2021-07-09T18:56:55.751977Z","iopub.status.idle":"2021-07-09T18:56:55.759357Z","shell.execute_reply.started":"2021-07-09T18:56:55.751945Z","shell.execute_reply":"2021-07-09T18:56:55.758125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport random\nfrom sklearn.utils import shuffle","metadata":{"execution":{"iopub.status.busy":"2021-07-09T18:57:03.974035Z","iopub.execute_input":"2021-07-09T18:57:03.974465Z","iopub.status.idle":"2021-07-09T18:57:04.197229Z","shell.execute_reply.started":"2021-07-09T18:57:03.974427Z","shell.execute_reply":"2021-07-09T18:57:04.196181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readImage(path):\n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    return rgb_img","metadata":{"execution":{"iopub.status.busy":"2021-07-09T18:57:18.57887Z","iopub.execute_input":"2021-07-09T18:57:18.579264Z","iopub.status.idle":"2021-07-09T18:57:18.584916Z","shell.execute_reply.started":"2021-07-09T18:57:18.579227Z","shell.execute_reply":"2021-07-09T18:57:18.583816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# random sampling\nshuffled_data = shuffle(df_data)\n\nfig, ax = plt.subplots(2,5, figsize=(20,8))\nfig.suptitle('Histopathologic scans of lymph node sections',fontsize=20)\n# Negatives\nfor i in range(5):\n#     path = os.path.join(train_path, idx)\n    ax[0,i].imshow(readImage(df_0['path'][i]))\n    # Create a Rectangle patch\n    box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',facecolor='none', linestyle=':', capstyle='round')\n    ax[0,i].add_patch(box)\nax[0,0].set_ylabel('Negative samples', size='large')\n# Positives\nfor i in range(5):\n#     path = os.path.join(train_path, idx)\n    ax[1,i].imshow(readImage(df_1['path'][i]))\n    # Create a Rectangle patch\n    box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='r',facecolor='none', linestyle=':', capstyle='round')\n    ax[1,i].add_patch(box)\nax[1,0].set_ylabel('Tumor tissue samples', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-09T19:03:07.485636Z","iopub.execute_input":"2021-07-09T19:03:07.486065Z","iopub.status.idle":"2021-07-09T19:03:09.178627Z","shell.execute_reply.started":"2021-07-09T19:03:07.486026Z","shell.execute_reply":"2021-07-09T19:03:09.177579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.image import crop_to_bounding_box","metadata":{"execution":{"iopub.status.busy":"2021-07-09T19:28:12.905532Z","iopub.execute_input":"2021-07-09T19:28:12.906077Z","iopub.status.idle":"2021-07-09T19:28:12.912173Z","shell.execute_reply.started":"2021-07-09T19:28:12.906029Z","shell.execute_reply":"2021-07-09T19:28:12.911218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cropped_image_tensor = crop_to_bounding_box(box, 32, 32, 32, 32)\nax[0,0].imshow(cropped_image_tensor)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T19:28:16.470252Z","iopub.execute_input":"2021-07-09T19:28:16.470653Z","iopub.status.idle":"2021-07-09T19:28:16.549931Z","shell.execute_reply.started":"2021-07-09T19:28:16.470619Z","shell.execute_reply":"2021-07-09T19:28:16.547955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2021-07-09T18:13:15.299394Z","iopub.execute_input":"2021-07-09T18:13:15.299803Z","iopub.status.idle":"2021-07-09T18:13:15.306723Z","shell.execute_reply.started":"2021-07-09T18:13:15.29977Z","shell.execute_reply":"2021-07-09T18:13:15.305435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 96\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 32\nval_batch_size = 32\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)\n\ndatagen = ImageDataGenerator(preprocessing_function=lambda x:(x - x.mean()) / x.std() if x.std() > 0 else x,\n                            horizontal_flip=True,\n                            vertical_flip=True)\n\ntrain_gen = datagen.flow_from_directory(train_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=train_batch_size,\n                                        class_mode='binary')\n\nval_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=val_batch_size,\n                                        class_mode='binary')\n\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=1,\n                                        class_mode='binary',\n                                        shuffle=False)","metadata":{},"execution_count":null,"outputs":[]}]}