{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport matplotlib as mpl\nimport random\nfrom sklearn.utils import shuffle\nfrom tqdm import tqdm_notebook\n\ndata = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')\ntrain_path = '../input/histopathologic-cancer-detection/train/'\ntest_path = '../input/histopathologic-cancer-detection/test/'\n\ndata['label'].value_counts()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-17T12:42:55.180114Z","iopub.execute_input":"2021-07-17T12:42:55.18069Z","iopub.status.idle":"2021-07-17T12:42:56.823115Z","shell.execute_reply.started":"2021-07-17T12:42:55.180604Z","shell.execute_reply":"2021-07-17T12:42:56.821999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking Null Values","metadata":{}},{"cell_type":"code","source":"data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T13:12:24.923119Z","iopub.execute_input":"2021-07-17T13:12:24.92402Z","iopub.status.idle":"2021-07-17T13:12:24.96209Z","shell.execute_reply.started":"2021-07-17T13:12:24.923963Z","shell.execute_reply":"2021-07-17T13:12:24.960834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:43:04.537259Z","iopub.execute_input":"2021-07-17T12:43:04.537671Z","iopub.status.idle":"2021-07-17T12:43:04.543106Z","shell.execute_reply.started":"2021-07-17T12:43:04.537635Z","shell.execute_reply":"2021-07-17T12:43:04.542383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = data['label'].value_counts()\nfreq = data['label'].value_counts(normalize = True)\n\ny_freq = pd.DataFrame({'label': list(map(str, count.index)),\n                       'count': count,\n                       'freq': freq})\n\ny_freq","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:43:08.285452Z","iopub.execute_input":"2021-07-17T12:43:08.286064Z","iopub.status.idle":"2021-07-17T12:43:08.328895Z","shell.execute_reply.started":"2021-07-17T12:43:08.286025Z","shell.execute_reply":"2021-07-17T12:43:08.328125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplot()\nplt.bar(y_freq.loc[:, 'label'],\n        y_freq.loc[:, 'freq'])\nplt.xlabel('labels')\nplt.ylabel('Relative frequency')\nplt.title('Labels distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:43:13.220461Z","iopub.execute_input":"2021-07-17T12:43:13.22103Z","iopub.status.idle":"2021-07-17T12:43:13.384062Z","shell.execute_reply.started":"2021-07-17T12:43:13.220979Z","shell.execute_reply":"2021-07-17T12:43:13.383223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readImage(path):\n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    return rgb_img / 255","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:43:37.156447Z","iopub.execute_input":"2021-07-17T12:43:37.157043Z","iopub.status.idle":"2021-07-17T12:43:37.162001Z","shell.execute_reply.started":"2021-07-17T12:43:37.156943Z","shell.execute_reply":"2021-07-17T12:43:37.161141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# random sampling\nshuffled_data = shuffle(data)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:43:40.041213Z","iopub.execute_input":"2021-07-17T12:43:40.041738Z","iopub.status.idle":"2021-07-17T12:43:40.084074Z","shell.execute_reply.started":"2021-07-17T12:43:40.041705Z","shell.execute_reply":"2021-07-17T12:43:40.083199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fig, ax = plt.subplots(2,5, figsize=(20,8))\nfig, ax = plt.subplots(4,5, figsize=(20,16))\nfig.suptitle('Histopathologic Cancer Detection',fontsize=20)\n# Negatives\n\nfor j in [0,1]:\n    for i, idx in enumerate(data[data['label'] == 0]['id'][(j*5):((j+1)*5)]):\n        path = os.path.join(train_path, idx)\n        ax[j,i].imshow(readImage(path + '.tif'))\n        # Create a Rectangle patch\n        # box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='g',\n        #                    facecolor='none', linestyle=':', capstyle='round')\n        box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',\n                            facecolor='none', linestyle=':', capstyle='round')\n        ax[j,i].add_patch(box)\n    ax[j,0].set_ylabel('Negative samples', size='large')\n\nfor j in [0,1]:\n    # Positives\n    for i, idx in enumerate(data[data['label'] == 1]['id'][(j*5):((j+1)*5)]):\n        path = os.path.join(train_path, idx)\n        ax[(j+2),i].imshow(readImage(path + '.tif'))\n        # Create a Rectangle patch\n        box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='r',\n                            facecolor='none', linestyle=':', capstyle='round')\n        ax[(j+2),i].add_patch(box)\n    ax[(j+2),0].set_ylabel('Tumor tissue samples', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:50:03.972827Z","iopub.execute_input":"2021-07-17T12:50:03.973218Z","iopub.status.idle":"2021-07-17T12:50:07.417258Z","shell.execute_reply.started":"2021-07-17T12:50:03.973187Z","shell.execute_reply":"2021-07-17T12:50:07.415889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nORIGINAL_SIZE = 96\n\n# AUGMENTATION VARIABLES\nCROP_SIZE = 90\n# CROP_SIZE = 80\n# CROP_SIZE = 66 \n# CROP_SIZE = 60\n# CROP_SIZE = 48\nRANDOM_ROTATION = 3\n# RANDOM_ROTATION = 30\n# RANDOM_ROTATION = 45\nRANDOM_SHIFT = 2 \n# RANDOM_SHIFT = 8\nRANDOM_BRIGHTNESS = 7\nRANDOM_CONTRAST = 5\nRANDOM_90_DEG_TURN = 1","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:51:51.858864Z","iopub.execute_input":"2021-07-17T12:51:51.859218Z","iopub.status.idle":"2021-07-17T12:51:51.864054Z","shell.execute_reply.started":"2021-07-17T12:51:51.859188Z","shell.execute_reply":"2021-07-17T12:51:51.863323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readCroppedImage(path, augmentations = True):\n    # augmentations parameter is included for counting statistics from images, where we don't want augmentations\n    \n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    \n    if(not augmentations):\n        return rgb_img / 255\n    \n    # Random flip\n    flip_hor = bool(random.getrandbits(1))\n    flip_ver = bool(random.getrandbits(1))\n    if(flip_hor):\n        rgb_img = rgb_img[:, ::-1]\n    if(flip_ver):\n        rgb_img = rgb_img[::-1, :]\n        \n    # random rotation\n    # Clockwise rotation\n    rotation = random.randint(-RANDOM_ROTATION,RANDOM_ROTATION)\n    if(RANDOM_90_DEG_TURN == 1):\n        rotation += random.randint(-1,1) * 90\n    M = cv2.getRotationMatrix2D((48,48),rotation,1)   # the center point is the rotation anchor\n    rgb_img = cv2.warpAffine(rgb_img,M,(96,96))\n    \n    #random x,y-shift\n    x = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT) # Horizontal shift to the right\n    y = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT) # Vertical shift to the bottom\n    \n    # crop to center and normalize to 0-1 range\n    start_crop = (ORIGINAL_SIZE - CROP_SIZE) // 2\n    end_crop = start_crop + CROP_SIZE\n    # rgb_img = rgb_img[(start_crop + x):(end_crop + x), (start_crop + y):(end_crop + y)] / 255\n    rgb_img = rgb_img[(start_crop - y):(end_crop - y), (start_crop - x):(end_crop - x)] / 255\n    \n    # Random brightness\n    br = random.randint(-RANDOM_BRIGHTNESS, RANDOM_BRIGHTNESS) / 100.\n    rgb_img = rgb_img + br\n    \n    # Random contrast\n    cr = 1.0 + random.randint(-RANDOM_CONTRAST, RANDOM_CONTRAST) / 100.\n    rgb_img = rgb_img * cr\n    \n    # clip values to 0-1 range\n    rgb_img = np.clip(rgb_img, 0, 1.0)\n    \n    return rgb_img","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:51:07.296266Z","iopub.execute_input":"2021-07-17T12:51:07.296808Z","iopub.status.idle":"2021-07-17T12:51:07.31105Z","shell.execute_reply.started":"2021-07-17T12:51:07.296763Z","shell.execute_reply":"2021-07-17T12:51:07.310011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readCroppedImage2(path, augmentations = True):\n    # augmentations parameter is included for counting statistics from images, where we don't want augmentations\n    \n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    \n    if(not augmentations):\n        return rgb_img / 255, 0, 0, 0\n    \n    # Random flip\n    flip_hor = bool(random.getrandbits(1))\n    flip_ver = bool(random.getrandbits(1))\n    if(flip_hor):\n        rgb_img = rgb_img[:, ::-1]\n    if(flip_ver):\n        rgb_img = rgb_img[::-1, :]\n    \n    # random rotation\n    # Clockwise rotation\n    rotation = random.randint(-RANDOM_ROTATION,RANDOM_ROTATION)\n    if(RANDOM_90_DEG_TURN == 1):\n        rotation += random.randint(-1,1) * 90\n    M = cv2.getRotationMatrix2D((48,48),rotation,1)   # the center point is the rotation anchor\n    rgb_img = cv2.warpAffine(rgb_img,M,(96,96))\n    \n    #random x,y-shift\n    x = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT) # Horizontal shift to the right \n    y = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT) # Vertical shift to the bottom\n    \n    # crop to center and normalize to 0-1 range\n    start_crop = (ORIGINAL_SIZE - CROP_SIZE) // 2\n    end_crop = start_crop + CROP_SIZE\n    rgb_img = rgb_img[(start_crop - y):(end_crop - y), (start_crop - x):(end_crop - x)] / 255\n       \n        \n    # Random brightness\n    br = random.randint(-RANDOM_BRIGHTNESS, RANDOM_BRIGHTNESS) / 100.\n    rgb_img = rgb_img + br\n    \n    # Random contrast\n    cr = 1.0 + random.randint(-RANDOM_CONTRAST, RANDOM_CONTRAST) / 100.\n    rgb_img = rgb_img * cr\n    \n    # clip values to 0-1 range\n    rgb_img = np.clip(rgb_img, 0, 1.0)\n    \n    return rgb_img, rotation, x, y","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:51:12.919761Z","iopub.execute_input":"2021-07-17T12:51:12.920125Z","iopub.status.idle":"2021-07-17T12:51:12.932307Z","shell.execute_reply.started":"2021-07-17T12:51:12.920094Z","shell.execute_reply":"2021-07-17T12:51:12.931223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prove matplotlib\n\n# import numpy as np\n# import matplotlib.pyplot as plt\n# import matplotlib.patches as patches\n# import matplotlib as mpl\n\n# fig = plt.figure()\n# ax = fig.add_subplot(111)\n\nfig, ax = plt.subplots(1,1,figsize=(3,3))\nfig.subplots_adjust(hspace=0,wspace=0)\n\nts = ax.transData\ncoords = [48, 48]\ntr = mpl.transforms.Affine2D().rotate_deg_around(x=coords[0], y=coords[1], degrees=-30)\ntt = mpl.transforms.Affine2D().translate(20, -10)\n# tr = mpl.transforms.Affine2D().rotate_deg(degrees=-30)\nt = tt + tr + ts\n\nbox_1 = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='g',\n                          facecolor='none', linestyle=':', capstyle='round')\n\nax.add_patch(box_1)\n\nbox_2 = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',\n                          facecolor='none', linestyle=':', capstyle='round',\n                          transform=t)\n\nax.add_patch(box_2)\n\nplt.plot(coords[0], coords[1], marker='o', markersize=10, color=\"r\")\n\nplt.grid(True)\n\nax.set_xlim(0,96)\nax.set_ylim(0,96)\n\nplt.title('Original image')\n\nplt.ylabel('Negative samples', size='large')\n\npath = os.path.join(train_path, data[data['label'] == 0].loc[0,'id'])\nprint(path)\n\nplt.imshow(readImage(path + '.tif'))\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:51:18.65854Z","iopub.execute_input":"2021-07-17T12:51:18.658885Z","iopub.status.idle":"2021-07-17T12:51:18.859981Z","shell.execute_reply.started":"2021-07-17T12:51:18.658853Z","shell.execute_reply":"2021-07-17T12:51:18.858948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read manually the image\npath = os.path.join(train_path, data[data['label'] == 0].loc[0,'id']) + '.tif'\nprint(path)\n\n# OpenCV reads the image in bgr format by default\nbgr_img = cv2.imread(path)\n\n# We flip it to rgb for visualization purposes\nb,g,r = cv2.split(bgr_img)\nrgb_img = cv2.merge([r,g,b])\n\nfig, ax = plt.subplots(1,2, figsize=(8,16))\nax[0].imshow(rgb_img)\nax[0].grid(True)\n\n## Random flip\n#flip_hor = bool(random.getrandbits(1))\n#flip_ver = bool(random.getrandbits(1))\n#if(flip_hor):\n#    rgb_img = rgb_img[:, ::-1]\n#if(flip_ver):\n#    rgb_img = rgb_img[::-1, :]\n    \n# random rotation\n# rotation = random.randint(-RANDOM_ROTATION,RANDOM_ROTATION)\n# if(RANDOM_90_DEG_TURN == 1):\n#    rotation += random.randint(-1,1) * 90\n\n# Clockwise rotation\n# rotation = 135\nrotation = RANDOM_ROTATION\n\nM = cv2.getRotationMatrix2D((48,48),rotation,1)   # the center point is the rotation anchor\nrgb_img = cv2.warpAffine(rgb_img,M,(96,96))\n\n#random x,y-shift\n# x = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT)\n# y = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT)\n# x = 8  # Horizontal shift to the right\nx = RANDOM_SHIFT  # Horizontal shift to the right\ny = 0             # Vertical shift to the bottom\n# RANDOM_SHIFT must be between 0 and (ORIGINAL_SIZE - CROP_SIZE) // 2\n\n# crop to center and normalize to 0-1 range\nstart_crop = (ORIGINAL_SIZE - CROP_SIZE) // 2\nend_crop = start_crop + CROP_SIZE\nrgb_img = rgb_img[(start_crop - y):(end_crop - y), (start_crop - x):(end_crop - x)] / 255\n   \n## Random brightness\n#br = random.randint(-RANDOM_BRIGHTNESS, RANDOM_BRIGHTNESS) / 100.\n#rgb_img = rgb_img + br\n\n## Random contrast\n#cr = 1.0 + random.randint(-RANDOM_CONTRAST, RANDOM_CONTRAST) / 100.\n#rgb_img = rgb_img * cr\n\n## clip values to 0-1 range\n#rgb_img = np.clip(rgb_img, 0, 1.0)\n\nax[1].imshow(rgb_img)\n\nts = ax[1].transData\ntr = mpl.transforms.Affine2D().rotate_deg_around(x = CROP_SIZE/2, y = CROP_SIZE/2, degrees = -rotation)\ntt = mpl.transforms.Affine2D().translate(x, y)\nt = tr + tt + ts\n\nbox = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n                         32,32,linewidth=4,edgecolor='r',\n                         facecolor='none', linestyle=':', capstyle='round',\n                         transform = t)\nax[1].add_patch(box)\nax[1].grid(True)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:51:59.36325Z","iopub.execute_input":"2021-07-17T12:51:59.363797Z","iopub.status.idle":"2021-07-17T12:51:59.770716Z","shell.execute_reply.started":"2021-07-17T12:51:59.363749Z","shell.execute_reply":"2021-07-17T12:51:59.769668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Nota: ci sono degli errori nelle traslazioni\n# non capisco come risolverli\n\nfig, ax = plt.subplots(2,5, figsize=(20,8))\nfig.suptitle('Cropped histopathologic scans of lymph node sections',fontsize=20)\n# Negatives\nfor i, idx in enumerate(shuffled_data[shuffled_data['label'] == 0]['id'][:5]):\n    path = os.path.join(train_path, idx)\n    img, rotation, x_shift, y_shift = readCroppedImage2(path + '.tif')\n    ax[0,i].imshow(img)\n       \n    ts = ax[0,i].transData\n    tr = mpl.transforms.Affine2D().rotate_deg_around(x = CROP_SIZE/2, y = CROP_SIZE/2, degrees = -rotation)\n    tt = mpl.transforms.Affine2D().translate(x_shift, y_shift)\n    t = tr + tt + ts\n    # box = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n    #                         32,32,linewidth=4,edgecolor='g',\n    #                         facecolor='none', linestyle=':', capstyle='round',\n    #                         transform = t)\n    box = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n                            32,32,linewidth=4,edgecolor='b',\n                            facecolor='none', linestyle=':', capstyle='round',\n                            transform = t)\n    ax[0,i].add_patch(box)        \n    \nax[0,0].set_ylabel('Negative samples', size='large')\n\n# Positives\nfor i, idx in enumerate(shuffled_data[shuffled_data['label'] == 1]['id'][:5]):\n    path = os.path.join(train_path, idx)\n    img, rotation, x_shift, y_shift = readCroppedImage2(path + '.tif')\n    ax[1,i].imshow(img)\n     \n    ts = ax[1,i].transData\n    tr = mpl.transforms.Affine2D().rotate_deg_around(x = CROP_SIZE/2, y = CROP_SIZE/2, degrees = -rotation)\n    tt = mpl.transforms.Affine2D().translate(x_shift, y_shift)\n    t = tr + tt + ts\n    box = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n                            32,32,linewidth=4,edgecolor='r',\n                            facecolor='none', linestyle=':', capstyle='round',\n                            transform = t)\n    ax[1,i].add_patch(box)\n    \nax[1,0].set_ylabel('Tumor tissue samples', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:52:12.654701Z","iopub.execute_input":"2021-07-17T12:52:12.655202Z","iopub.status.idle":"2021-07-17T12:52:14.775416Z","shell.execute_reply.started":"2021-07-17T12:52:12.655157Z","shell.execute_reply":"2021-07-17T12:52:14.774272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See augmentation effect\nfig, ax = plt.subplots(2,5, figsize=(30,12))\nfig.suptitle('Random augmentations to the same image',fontsize=20)\n# Negatives\n# for i, idx in enumerate(shuffled_data[shuffled_data['label'] == 0]['id'][:1]):\nfor i, idx in enumerate(shuffled_data[shuffled_data['label'] == 0]['id'][1:2]):\n    path = os.path.join(train_path, idx)\n    ax[0,0].imshow(readImage(path + '.tif'))\n    ax[0,0].set_title('Original image')\n    # box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='g',\n    box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='b',\n                            facecolor='none', linestyle=':', capstyle='round')\n    ax[0,0].add_patch(box)\n    for j in range(1,5):\n        img, rotation, x_shift, y_shift = readCroppedImage2(path + '.tif')\n        ax[0,j].imshow(img)\n       \n        ts = ax[0,j].transData\n        tr = mpl.transforms.Affine2D().rotate_deg_around(x = CROP_SIZE/2, y = CROP_SIZE/2, degrees = -rotation)\n        tt = mpl.transforms.Affine2D().translate(x_shift, y_shift)\n        t = tr + tt + ts\n        \n        box = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n                                # 32,32,linewidth=4,edgecolor='g',\n                                32,32,linewidth=4,edgecolor='b',\n                                facecolor='none', linestyle=':', capstyle='round',\n                                transform = t)\n        \n        ax[0,j].add_patch(box)\n        \nax[0,0].set_ylabel('Negative samples', size='large')\n        \nfor i, idx in enumerate(shuffled_data[shuffled_data['label'] == 1]['id'][:1]):\n    path = os.path.join(train_path, idx)\n    \n    ax[1,0].imshow(readImage(path + '.tif'))\n    ax[1,0].set_title('Original image')\n    box = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='r',\n                            facecolor='none', linestyle=':', capstyle='round')\n    ax[1,0].add_patch(box)\n    \n    for j in range(1,5):\n        img, rotation, x_shift, y_shift = readCroppedImage2(path + '.tif')\n        ax[1,j].imshow(img)\n         \n        ts = ax[1,j].transData\n        tr = mpl.transforms.Affine2D().rotate_deg_around(x = CROP_SIZE/2, y = CROP_SIZE/2, degrees = -rotation)\n        tt = mpl.transforms.Affine2D().translate(x_shift, y_shift)\n        t = tr + tt + ts\n        box = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n                                32,32,linewidth=4,edgecolor='r',\n                                facecolor='none', linestyle=':', capstyle='round',\n                                transform = t)\n        ax[1,j].add_patch(box)\n    #     ax[1,j].imshow(readCroppedImage(path + '.tif'))\n    #     ax[1,j].set_title('Cropped image')\n    #     box = patches.Rectangle((32-(ORIGINAL_SIZE - CROP_SIZE)/2,32-(ORIGINAL_SIZE - CROP_SIZE)/2),\n    #                             32,32,linewidth=4,edgecolor='r',\n    #                         facecolor='none', linestyle=':', capstyle='round')\n    #     ax[1,j].add_patch(box)\nax[1,0].set_ylabel('Tumor tissue samples', size='large')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T12:52:39.406259Z","iopub.execute_input":"2021-07-17T12:52:39.406636Z","iopub.status.idle":"2021-07-17T12:52:41.538966Z","shell.execute_reply.started":"2021-07-17T12:52:39.406605Z","shell.execute_reply":"2021-07-17T12:52:41.538116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# As we count the statistics, we can check if there are any completely black or white images\ndark_th = 10 / 255      # If no pixel reaches this threshold, image is considered too dark\nbright_th = 245 / 255   # If no pixel is under this threshold, image is considerd too bright\ntoo_dark_idx = []\ntoo_bright_idx = []\n\nx_tot = np.zeros(3)\nx2_tot = np.zeros(3)\ncounted_ones = 0\nfor i, idx in tqdm_notebook(enumerate(data['id']), 'computing statistics...(220025 it total)'):\n    path = os.path.join(train_path, idx)\n    imagearray = readCroppedImage(path + '.tif', augmentations = False).reshape(-1,3)\n    # is this too dark\n    if(imagearray.max() < dark_th):\n        too_dark_idx.append(idx)\n        continue # do not include in statistics\n    # is this too bright\n    if(imagearray.min() > bright_th):\n        too_bright_idx.append(idx)\n        continue # do not include in statistics\n    x_tot += imagearray.mean(axis=0)\n    x2_tot += (imagearray**2).mean(axis=0)\n    counted_ones += 1\n    \nchannel_avr = x_tot/counted_ones\nchannel_std = np.sqrt(x2_tot/counted_ones - channel_avr**2)\nchannel_avr,channel_std","metadata":{"execution":{"iopub.status.busy":"2021-07-15T19:34:54.759808Z","iopub.execute_input":"2021-07-15T19:34:54.760175Z","iopub.status.idle":"2021-07-15T20:08:46.237958Z","shell.execute_reply.started":"2021-07-15T19:34:54.760145Z","shell.execute_reply":"2021-07-15T20:08:46.236809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('There was {0} extremely dark image'.format(len(too_dark_idx)))\nprint('and {0} extremely bright images'.format(len(too_bright_idx)))\nprint('Dark one:')\nprint(too_dark_idx)\nprint('Bright ones:')\nprint(too_bright_idx)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:10:48.68786Z","iopub.execute_input":"2021-07-15T20:10:48.68836Z","iopub.status.idle":"2021-07-15T20:10:48.694737Z","shell.execute_reply.started":"2021-07-15T20:10:48.688301Z","shell.execute_reply":"2021-07-15T20:10:48.694017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data['id'].isin(too_dark_idx)]","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:11:03.460017Z","iopub.execute_input":"2021-07-15T20:11:03.460373Z","iopub.status.idle":"2021-07-15T20:11:03.489618Z","shell.execute_reply.started":"2021-07-15T20:11:03.460338Z","shell.execute_reply":"2021-07-15T20:11:03.488418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data['id'].isin(too_bright_idx)]","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:24:50.645498Z","iopub.execute_input":"2021-07-15T20:24:50.645958Z","iopub.status.idle":"2021-07-15T20:24:50.673361Z","shell.execute_reply.started":"2021-07-15T20:24:50.645908Z","shell.execute_reply":"2021-07-15T20:24:50.672453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,6, figsize=(25,9))\nfig.suptitle('Almost completely black or white images',fontsize=20)\n# Too dark\ni = 0\nfor idx in np.asarray(too_dark_idx)[:min(6, len(too_dark_idx))]:\n    lbl = data[data['id'] == idx]['label'].values[0]\n    path = os.path.join(train_path, idx)\n    ax[0,i].imshow(readCroppedImage(path + '.tif', augmentations = False))\n    ax[0,i].set_title(idx + '\\n label=' + str(lbl), fontsize = 8)\n    i += 1\nax[0,0].set_ylabel('Extremely dark images', size='large')\nfor j in range(min(6, len(too_dark_idx)), 6):\n    ax[0,j].axis('off') # hide axes if there are less than 6\n# Too bright\ni = 0\nfor idx in np.asarray(too_bright_idx)[:min(6, len(too_bright_idx))]:\n    lbl = data[data['id'] == idx]['label'].values[0]\n    path = os.path.join(train_path, idx)\n    ax[1,i].imshow(readCroppedImage(path + '.tif', augmentations = False))\n    ax[1,i].set_title(idx + '\\n label=' + str(lbl), fontsize = 8)\n    i += 1\nax[1,0].set_ylabel('Extremely bright images', size='large')\nfor j in range(min(6, len(too_bright_idx)), 6):\n    ax[1,j].axis('off') # hide axes if there are less than 6","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:25:13.99612Z","iopub.execute_input":"2021-07-15T20:25:13.996448Z","iopub.status.idle":"2021-07-15T20:25:15.380438Z","shell.execute_reply.started":"2021-07-15T20:25:13.996418Z","shell.execute_reply":"2021-07-15T20:25:15.379436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# we read the csv file earlier to pandas dataframe, now we set index to id so we can perform\ntrain_df = data.set_index('id')\n\n#If removing outliers, uncomment the four lines below\n#print('Before removing outliers we had {0} training samples.'.format(train_df.shape[0]))\n#train_df = train_df.drop(labels=too_dark_idx, axis=0)\n#train_df = train_df.drop(labels=too_bright_idx, axis=0)\n#print('After removing outliers we have {0} training samples.'.format(train_df.shape[0]))\n\ntrain_names = train_df.index.values\ntrain_labels = np.asarray(train_df['label'].values)\n\n# split, this function returns more than we need as we only need the validation indexes for fastai\n# tr_n, tr_idx, val_n, val_idx = train_test_split(train_names, range(len(train_names)), test_size=0.1, stratify=train_labels, random_state=123)\n# tr_n, val_n, tr_idx, val_idx = train_test_split(train_names, range(len(train_names)),\n#                                                 test_size=0.1, stratify=train_labels, random_state=123)\ntr_n, val_n, tr_idx, val_idx = train_test_split(train_names, range(len(train_names)),\n                                                test_size=0.1, stratify=train_labels, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:25:32.316436Z","iopub.execute_input":"2021-07-15T20:25:32.316753Z","iopub.status.idle":"2021-07-15T20:25:32.61498Z","shell.execute_reply.started":"2021-07-15T20:25:32.316724Z","shell.execute_reply":"2021-07-15T20:25:32.614126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.shape)\n\nprint(tr_n.shape)\nprint(len(tr_idx))\n\nprint(val_n.shape)\nprint(len(val_idx))\n\n\n# print(tr_n)\n# print(data.iloc[tr_idx,:])\n\n# print(val_n)\n# print(data.iloc[val_idx,:])\n\nprint(type(val_idx))","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:25:46.930207Z","iopub.execute_input":"2021-07-15T20:25:46.93059Z","iopub.status.idle":"2021-07-15T20:25:46.936588Z","shell.execute_reply.started":"2021-07-15T20:25:46.930555Z","shell.execute_reply":"2021-07-15T20:25:46.935641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fastai 1.0\nfrom fastai import *\nfrom fastai.vision import *\nfrom torchvision.models import *    # import *=all the models from torchvision  \n\narch = densenet169                  # specify model architecture, densenet169 seems to perform well for this data but you could experiment\nBATCH_SIZE = 128                    # specify batch size, hardware restrics this one. Large batch sizes may run out of GPU memory\nsz = CROP_SIZE                      # input size is the crop size\nMODEL_PATH = str(arch).split()[1]   # this will extrat the model name as the model file name e.g. 'resnet50'","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:26:03.965245Z","iopub.execute_input":"2021-07-15T20:26:03.965624Z","iopub.status.idle":"2021-07-15T20:26:05.274423Z","shell.execute_reply.started":"2021-07-15T20:26:03.965591Z","shell.execute_reply":"2021-07-15T20:26:05.273564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/input\nfrom fastaiv1 import fastai\nfrom fastaiv1 import *\n\n%cd /kaggle/","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:42:22.445555Z","iopub.execute_input":"2021-07-15T20:42:22.445892Z","iopub.status.idle":"2021-07-15T20:42:22.453515Z","shell.execute_reply.started":"2021-07-15T20:42:22.445861Z","shell.execute_reply":"2021-07-15T20:42:22.452467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/input\nfrom fastaiv1.fastai.vision import *\n%cd /kaggle/","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:42:50.625703Z","iopub.execute_input":"2021-07-15T20:42:50.626048Z","iopub.status.idle":"2021-07-15T20:42:51.74386Z","shell.execute_reply.started":"2021-07-15T20:42:50.626017Z","shell.execute_reply":"2021-07-15T20:42:51.742564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import fastai\nfastai.__version__","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:40:50.564809Z","iopub.execute_input":"2021-07-15T20:40:50.565368Z","iopub.status.idle":"2021-07-15T20:40:50.571914Z","shell.execute_reply.started":"2021-07-15T20:40:50.565314Z","shell.execute_reply":"2021-07-15T20:40:50.571021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vision.__version__","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:43:03.080469Z","iopub.execute_input":"2021-07-15T20:43:03.0808Z","iopub.status.idle":"2021-07-15T20:43:03.085472Z","shell.execute_reply.started":"2021-07-15T20:43:03.080769Z","shell.execute_reply":"2021-07-15T20:43:03.084875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(str(arch))\n\nprint(str(arch).split())\n\nprint(str(arch).split()[1])","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:26:14.770696Z","iopub.execute_input":"2021-07-15T20:26:14.771061Z","iopub.status.idle":"2021-07-15T20:26:14.776896Z","shell.execute_reply.started":"2021-07-15T20:26:14.771025Z","shell.execute_reply":"2021-07-15T20:26:14.775631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataframe for the fastai loader\ntrain_dict = {'name': train_path + train_names, 'label': train_labels}\ndf = pd.DataFrame(data=train_dict)\n# create test dataframe\ntest_names = []\nfor f in os.listdir(test_path):\n    test_names.append(test_path + f)\ndf_test = pd.DataFrame(np.asarray(test_names), columns=['name'])","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:26:28.030744Z","iopub.execute_input":"2021-07-15T20:26:28.031242Z","iopub.status.idle":"2021-07-15T20:26:29.476185Z","shell.execute_reply.started":"2021-07-15T20:26:28.031207Z","shell.execute_reply":"2021-07-15T20:26:29.47531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Subclass ImageList to use our own image opening function\nclass MyImageItemList(ImageList):\n    def open(self, fn:PathOrStr)->Image:\n        # print(fn)\n        img = readCroppedImage(fn.replace('/./','').replace('//','/'))\n        # img = readCroppedImage(fn.replace('/./','').replace('//',''))\n        # Decomment the previous line if you want to run locally the code\n        # This ndarray image has to be converted to tensor before passing on as fastai Image, we can use pil2tensor\n        return vision.Image(px=pil2tensor(img, np.float32))","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:43:09.100676Z","iopub.execute_input":"2021-07-15T20:43:09.101282Z","iopub.status.idle":"2021-07-15T20:43:09.105513Z","shell.execute_reply.started":"2021-07-15T20:43:09.101241Z","shell.execute_reply":"2021-07-15T20:43:09.104874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create ImageDataBunch using fastai data block API\nimgDataBunch = (MyImageItemList.from_df(path='/', df=df, suffix='.tif')\n        #Where to find the data?\n        .split_by_idx(val_idx)\n        #How to split in train/valid?\n        .label_from_df(cols='label')\n        #Where are the labels?\n        .add_test(MyImageItemList.from_df(path='/', df=df_test))\n        #dataframe pointing to the test set?\n        .transform(tfms=[[],[]], size=sz)\n        # We have our custom transformations implemented in the image loader but we could apply transformations also here\n        # Even though we don't apply transformations here, we set two empty lists to tfms. Train and Validation augmentations\n        .databunch(bs=BATCH_SIZE)\n        # convert to databunch\n        #.normalize([tensor([0.702447, 0.546243, 0.696453]), tensor([0.238893, 0.282094, 0.216251])])\n        .normalize([tensor(list(channel_avr)), tensor(list(channel_std))])\n        # Normalize with training set stats. These are means and std's of each three channel and we calculated these previously in the stats step.\n       )","metadata":{"execution":{"iopub.status.busy":"2021-07-15T20:43:29.635118Z","iopub.execute_input":"2021-07-15T20:43:29.635481Z","iopub.status.idle":"2021-07-15T20:43:30.757867Z","shell.execute_reply.started":"2021-07-15T20:43:29.635446Z","shell.execute_reply":"2021-07-15T20:43:30.755536Z"},"trusted":true},"execution_count":null,"outputs":[]}]}