{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-05-23T01:43:12.204055Z","iopub.execute_input":"2022-05-23T01:43:12.204678Z","iopub.status.idle":"2022-05-23T01:43:12.230934Z","shell.execute_reply.started":"2022-05-23T01:43:12.204588Z","shell.execute_reply":"2022-05-23T01:43:12.23027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:43:12.233321Z","iopub.execute_input":"2022-05-23T01:43:12.233787Z","iopub.status.idle":"2022-05-23T01:43:16.648283Z","shell.execute_reply.started":"2022-05-23T01:43:12.23375Z","shell.execute_reply":"2022-05-23T01:43:16.647472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nprint(keras.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:43:16.649459Z","iopub.execute_input":"2022-05-23T01:43:16.649973Z","iopub.status.idle":"2022-05-23T01:43:17.345922Z","shell.execute_reply.started":"2022-05-23T01:43:16.649933Z","shell.execute_reply":"2022-05-23T01:43:17.345214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimagePath='../input/human-protein-atlas-image-classification/train'\nobj = os.scandir(imagePath)\n# List all files and directories\n# in the specified path\nimagesFiles= []\nprint(\"Files and Directories in '% s':\" % imagePath)\nfor entry in obj :\n    if entry.is_file():\n        imagesFiles.append(entry.name)\n\nprint(len(imagesFiles))","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:43:17.347379Z","iopub.execute_input":"2022-05-23T01:43:17.347623Z","iopub.status.idle":"2022-05-23T01:45:30.598245Z","shell.execute_reply.started":"2022-05-23T01:43:17.347588Z","shell.execute_reply":"2022-05-23T01:45:30.597446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls '../input/human-protein-atlas-image-classification/train'| wc -l","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:30.600333Z","iopub.execute_input":"2022-05-23T01:45:30.601074Z","iopub.status.idle":"2022-05-23T01:45:31.738409Z","shell.execute_reply.started":"2022-05-23T01:45:30.601034Z","shell.execute_reply":"2022-05-23T01:45:31.737564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install https://github.com/CellProfiling/HPA-Cell-Segmentation/archive/master.zip","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:31.740245Z","iopub.execute_input":"2022-05-23T01:45:31.740527Z","iopub.status.idle":"2022-05-23T01:45:44.494654Z","shell.execute_reply.started":"2022-05-23T01:45:31.740481Z","shell.execute_reply":"2022-05-23T01:45:44.49384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nimport hpacellseg.cellsegmentator as cellsegmentator\nfrom hpacellseg.utils import label_cell, label_nuclei\nfrom imgaug import augmenters as iaa\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:44.496285Z","iopub.execute_input":"2022-05-23T01:45:44.496549Z","iopub.status.idle":"2022-05-23T01:45:47.491439Z","shell.execute_reply.started":"2022-05-23T01:45:44.496511Z","shell.execute_reply":"2022-05-23T01:45:47.490684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.read_csv('../input/human-protein-atlas-image-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:47.492687Z","iopub.execute_input":"2022-05-23T01:45:47.492935Z","iopub.status.idle":"2022-05-23T01:45:47.544563Z","shell.execute_reply.started":"2022-05-23T01:45:47.492901Z","shell.execute_reply":"2022-05-23T01:45:47.543881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain,_   = train_test_split(all_data, test_size = 0.5, random_state=42)\n\ntrain = train.reset_index()\ntrain['aug'] = 0\ntrain = train[['Id','Target','aug']]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:47.545735Z","iopub.execute_input":"2022-05-23T01:45:47.545977Z","iopub.status.idle":"2022-05-23T01:45:47.576565Z","shell.execute_reply.started":"2022-05-23T01:45:47.545943Z","shell.execute_reply":"2022-05-23T01:45:47.575822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:47.577789Z","iopub.execute_input":"2022-05-23T01:45:47.578032Z","iopub.status.idle":"2022-05-23T01:45:47.582813Z","shell.execute_reply.started":"2022-05-23T01:45:47.577984Z","shell.execute_reply":"2022-05-23T01:45:47.582146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\nlabels = [x.split() for x in train.Target.tolist()]\nlabels = list(itertools.chain(*labels))\nflat_labels =[int(l) for l in labels]\n\nflat_labels =np.array(flat_labels)\n(label, counts) = np.unique(flat_labels, return_counts=True)\n# list(zip(label, counts))\nlb_to_balanced = [1,2, 3, 4,5,6, 7  ,    8,    9,  10, 11 ,12, 13,14, 15, 16, 17,  18, 19 , 20, 21,22, 23,24,25,26,27]\nlb_to_added =    [5500,4300,5200,5100\n                  ,5000,5500,5000, 6000\n                  , 6000,6000 ,5500, 5700\n                  ,6000,5500 ,6000,5700 , 5900\n                  ,5500 ,5300,6000, 4200\n                  , 5700, 4800, 5900, 2000,5900, 6000]\n\n\ndef generate_large_list(ls, size):\n    n_copies = size // len(ls)\n    excess = size % len(ls)\n\n    result = sorted([element \n                     for i in range(n_copies) \n                     for element in ls] + ls[:excess]) \n    print(len(result))\n    return result\n\ndf = train.copy()\ndf['labels'] = [[ int(l) for l in x.split()] for x in df.Target.tolist()]\nminority_files = []\ntargets = []\nfor label in lb_to_balanced:\n    files = []\n    for i in range(len(df)):\n        if str(label) in df.iloc[i,1]:\n            files.append( df.iloc[i,0])\n    size = lb_to_added[lb_to_balanced.index(label)]\n    minority_files.extend( generate_large_list(files, size) )\n    targets.extend([str(label) for _ in range(size)])\n#     targets.extend(np.array([np.array([label]) for _ in range(size)]))\n\n\n# targets = np.array(targets)\nminority_dataframe = pd.DataFrame({'Id': minority_files, 'Target': targets})\nminority_dataframe['aug'] = 1\n\ntrain = train[['Id', 'Target', 'aug']]\ntrain = pd.concat([train, minority_dataframe], ignore_index=True)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:45:47.584343Z","iopub.execute_input":"2022-05-23T01:45:47.584853Z","iopub.status.idle":"2022-05-23T01:46:02.312973Z","shell.execute_reply.started":"2022-05-23T01:45:47.584816Z","shell.execute_reply":"2022-05-23T01:46:02.312192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def augment(image):\n    augment_img = iaa.Sequential([\n        iaa.OneOf([\n            iaa.Affine(rotate=0),\n            iaa.Affine(rotate=90),\n            iaa.Affine(rotate=180),\n            iaa.Affine(rotate=270),\n            iaa.Fliplr(0.5),\n            iaa.Flipud(0.5),\n        ])], random_order=True)\n\n    image_aug = augment_img.augment_image(image)\n    return image_aug","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:46:02.314167Z","iopub.execute_input":"2022-05-23T01:46:02.314495Z","iopub.status.idle":"2022-05-23T01:46:02.322086Z","shell.execute_reply.started":"2022-05-23T01:46:02.314458Z","shell.execute_reply":"2022-05-23T01:46:02.321146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input: list of image filters as png\n# Output: list of image filters as np.arrays\ndef image_to_arrays(path, aug):\n    \n    image_arrays = list()\n    for image in path:\n        array = np.asarray(Image.open(image))\n        if aug == 1:\n            array = augment(array)\n        image_arrays.append(array)\n        \n        \n    return image_arrays","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:46:02.325525Z","iopub.execute_input":"2022-05-23T01:46:02.325803Z","iopub.status.idle":"2022-05-23T01:46:02.333382Z","shell.execute_reply.started":"2022-05-23T01:46:02.325764Z","shell.execute_reply":"2022-05-23T01:46:02.332649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_blended_image(images): \n    # get rgby images for sample\n\n    # blend rgby images into single array\n    blended_array = np.stack(images[:-1], 2)\n\n    # Create PIL Image\n    blended_image = Image.fromarray( np.uint8(blended_array) )\n    return blended_image","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:46:02.334722Z","iopub.execute_input":"2022-05-23T01:46:02.335147Z","iopub.status.idle":"2022-05-23T01:46:02.342336Z","shell.execute_reply.started":"2022-05-23T01:46:02.335108Z","shell.execute_reply":"2022-05-23T01:46:02.341659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colours = ['_red.png', '_blue.png', '_yellow.png', '_green.png']\nTRAIN = '../input/human-protein-atlas-image-classification/train'\npaths = [[os.path.join(TRAIN, train.iloc[idx,0])+ colour for colour in colours] for idx in range(len(train))]\naugs = train.aug.values","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:46:02.343913Z","iopub.execute_input":"2022-05-23T01:46:02.344231Z","iopub.status.idle":"2022-05-23T01:46:23.624062Z","shell.execute_reply.started":"2022-05-23T01:46:02.344195Z","shell.execute_reply":"2022-05-23T01:46:23.623044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUC_MODEL = \"./nuclei-model.pth\"\nCELL_MODEL = \"./cell-model.pth\"\nsegmentator = cellsegmentator.CellSegmentator(NUC_MODEL,CELL_MODEL,scale_factor=0.25,device=\"cuda\",padding=True,multi_channel_model=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:46:23.626175Z","iopub.execute_input":"2022-05-23T01:46:23.626827Z","iopub.status.idle":"2022-05-23T01:47:55.887618Z","shell.execute_reply.started":"2022-05-23T01:46:23.626785Z","shell.execute_reply":"2022-05-23T01:47:55.886895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_counts = []\nlabels = []\nfor label in train['Target']:\n    sep = label.split(' ')\n    for num in sep:\n        labels.append(int(num))\ncounts = pd.value_counts(labels)\n\n# It's an ugly plot, but I'm trying to save some time here...\nplt.bar(x = counts.index,height=counts)\nplt.xticks(counts.index)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:47:55.888883Z","iopub.execute_input":"2022-05-23T01:47:55.889152Z","iopub.status.idle":"2022-05-23T01:47:56.337605Z","shell.execute_reply.started":"2022-05-23T01:47:55.889116Z","shell.execute_reply":"2022-05-23T01:47:56.336943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = paths[3]\naug = augs[3]\narrays = image_to_arrays(image, aug)\nnuclei = arrays[1]\ncell = arrays[:-1]\n# Nuclei segmentation\nnuc_segmentations = segmentator.pred_nuclei([nuclei])\nf, ax = plt.subplots(1, 2, figsize=(16,16))\nax[0].imshow(arrays[1])\nax[0].set_title('Original Nucleis', size=20)\nax[1].imshow(nuc_segmentations[0])\nax[1].set_title('Segmented Nucleis', size=20)\nplt.show()\n#\n\n# Nuclei segmentation\nnuc_segmentations = segmentator.pred_nuclei([nuclei])\n    \n# Cell segmentation\ninter_step = [[i] for i in image[:-1]]\ncell_segmentations = segmentator.pred_cells(inter_step)\n\nf, ax = plt.subplots(1, 2, figsize=(16,16))\nax[0].imshow(get_blended_image(arrays))\nax[0].set_title('Original Cells', size=20)\nax[1].imshow(cell_segmentations[0])\nax[1].set_title('Segmented Cells', size=20)\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:47:56.33895Z","iopub.execute_input":"2022-05-23T01:47:56.339219Z","iopub.status.idle":"2022-05-23T01:48:02.720386Z","shell.execute_reply.started":"2022-05-23T01:47:56.339183Z","shell.execute_reply":"2022-05-23T01:48:02.71907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def iou(target, prediction):\n    intersection = np.logical_and(target, prediction)\n    union = np.logical_or(target, prediction)\n    iou_score_segmented_nuclies = np.sum(intersection) / np.sum(union)\n    return iou_score_segmented_nuclies","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:48:02.721711Z","iopub.execute_input":"2022-05-23T01:48:02.724281Z","iopub.status.idle":"2022-05-23T01:48:02.729531Z","shell.execute_reply.started":"2022-05-23T01:48:02.724241Z","shell.execute_reply":"2022-05-23T01:48:02.728687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\ndef generate_images(paths):\n    for i in range(len(paths)):\n        image = paths[i]\n        aug = augs[i]\n        arrays = image_to_arrays(image, aug)\n        yield arrays\n\n\niouNucleiSegmentationForDataset = 0\niouCellSegmentationForDataset = 0\nindex = 1\nimages = generate_images(paths)    \nimages_size = len(paths)\n\n\nfor arrays in images:\n    #print(aug)\n    #arrays = image_to_arrays(image, aug)\n    nuclei = arrays[1]\n    cell = arrays[:-1]\n    # Nuclei segmentation\n    nuc_segmentations = segmentator.pred_nuclei([nuclei])\n    nuc_segmented = cv2.cvtColor(nuc_segmentations[0], cv2.COLOR_BGR2GRAY)\n\n    # Cell segmentation\n    inter_step = [[i] for i in image[:-1]]\n    cell_segmentations = segmentator.pred_cells(inter_step)\n    cell_segmented = cv2.cvtColor(cell_segmentations[0], cv2.COLOR_BGR2GRAY)\n    \n    cell_nuclei_mask, cell_mask = label_cell(nuc_segmentations[0], cell_segmentations[0])    \n    \n#     original_cell = np.array(get_blended_image(arrays))\n    \n   # iouNucleiSegmentationForDataset += iou(nuc_segmented, nuclei_mask)\n    iouCellSegmentationForDataset += iou(cell_segmented, cell_mask)\n    print(f'{index} out of {len(paths)}')\n    index += 1\n    #break\nprint(f'segmented nuclies for the dataset is {iouNucleiSegmentationForDataset/images_size}')\nprint(f'Cell segmentation {iouCellSegmentationForDataset/images_size}')","metadata":{"execution":{"iopub.status.busy":"2022-05-23T01:48:02.730843Z","iopub.execute_input":"2022-05-23T01:48:02.731567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_nuclei_mask, cell_mask = label_cell(nuc_segmentations[0], cell_segmentations[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.imshow(get_blended_image(arrays))\nplt.imshow(cell_mask, alpha=0.7)\nplt.title('Segmentation results', size=40)\nplt.axis('off')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#original_cell = np.array(get_blended_image(arrays))\n#print(f'Segmented nuclies iou value is {iou(original_cell, nuc_segmentations[0])} for the image')\n#print(f'Cell segmentation iou value is {iou(original_cell, cell_segmentations[0])} for the image')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}