{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-05-31T18:29:00.484206Z","iopub.execute_input":"2022-05-31T18:29:00.484752Z","iopub.status.idle":"2022-05-31T18:29:00.489583Z","shell.execute_reply.started":"2022-05-31T18:29:00.484714Z","shell.execute_reply":"2022-05-31T18:29:00.488783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.491568Z","iopub.execute_input":"2022-05-31T18:29:00.49214Z","iopub.status.idle":"2022-05-31T18:29:00.510677Z","shell.execute_reply.started":"2022-05-31T18:29:00.492079Z","shell.execute_reply":"2022-05-31T18:29:00.509647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nprint(keras.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.511886Z","iopub.execute_input":"2022-05-31T18:29:00.51235Z","iopub.status.idle":"2022-05-31T18:29:00.52442Z","shell.execute_reply.started":"2022-05-31T18:29:00.512318Z","shell.execute_reply":"2022-05-31T18:29:00.523541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimagePath='../input/human-protein-atlas-image-classification/train'\nobj = os.scandir(imagePath)\n# List all files and directories\n# in the specified path\nimagesFiles= []\nprint(\"Files and Directories in '% s':\" % imagePath)\nfor entry in obj :\n    if entry.is_file():\n        imagesFiles.append(entry.name)\n\nprint(len(imagesFiles))","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.525536Z","iopub.execute_input":"2022-05-31T18:29:00.525851Z","iopub.status.idle":"2022-05-31T18:29:00.558637Z","shell.execute_reply.started":"2022-05-31T18:29:00.525822Z","shell.execute_reply":"2022-05-31T18:29:00.557307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls '../input/human-protein-atlas-image-classification/train'| wc -l","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.559888Z","iopub.status.idle":"2022-05-31T18:29:00.560385Z","shell.execute_reply.started":"2022-05-31T18:29:00.560149Z","shell.execute_reply":"2022-05-31T18:29:00.560172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install \"../input/pycocotools202/pycocotools-2.0.2-cp37-cp37m-linux_x86_64.whl\"\n!pip install \"../input/hpapytorchzoozip/pytorch_zoo-master\"","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.561615Z","iopub.status.idle":"2022-05-31T18:29:00.561943Z","shell.execute_reply.started":"2022-05-31T18:29:00.561795Z","shell.execute_reply":"2022-05-31T18:29:00.561811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install https://github.com/CellProfiling/HPA-Cell-Segmentation/archive/master.zip","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.563091Z","iopub.status.idle":"2022-05-31T18:29:00.563473Z","shell.execute_reply.started":"2022-05-31T18:29:00.563314Z","shell.execute_reply":"2022-05-31T18:29:00.563333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nimport hpacellseg.cellsegmentator as cellsegmentator\nfrom hpacellseg.utils import label_cell, label_nuclei\nfrom imgaug import augmenters as iaa\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.565164Z","iopub.status.idle":"2022-05-31T18:29:00.565936Z","shell.execute_reply.started":"2022-05-31T18:29:00.565744Z","shell.execute_reply":"2022-05-31T18:29:00.565766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.read_csv('../input/human-protein-atlas-image-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.567217Z","iopub.status.idle":"2022-05-31T18:29:00.567564Z","shell.execute_reply.started":"2022-05-31T18:29:00.567408Z","shell.execute_reply":"2022-05-31T18:29:00.567425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain,_   = train_test_split(all_data, test_size = 0.5, random_state=42)\n\ntrain = train.reset_index()\ntrain['aug'] = 0\ntrain = train[['Id','Target','aug']]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.568708Z","iopub.status.idle":"2022-05-31T18:29:00.56905Z","shell.execute_reply.started":"2022-05-31T18:29:00.568895Z","shell.execute_reply":"2022-05-31T18:29:00.568913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.570864Z","iopub.status.idle":"2022-05-31T18:29:00.571401Z","shell.execute_reply.started":"2022-05-31T18:29:00.571202Z","shell.execute_reply":"2022-05-31T18:29:00.571229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\nlabels = [x.split() for x in train.Target.tolist()]\nlabels = list(itertools.chain(*labels))\nflat_labels =[int(l) for l in labels]\n\nflat_labels =np.array(flat_labels)\n(label, counts) = np.unique(flat_labels, return_counts=True)\n# list(zip(label, counts))\nlb_to_balanced = [1,2, 3, 4,5,6, 7  ,    8,    9,  10, 11 ,12, 13,14, 15, 16, 17,  18, 19 , 20, 21,22, 23,24,25,26,27]\nlb_to_added =    [5500,4300,5200,5100\n                  ,5000,5500,5000, 6000\n                  , 6000,6000 ,5500, 5700\n                  ,6000,5500 ,6000,5700 , 5900\n                  ,5500 ,5300,6000, 4200\n                  , 5700, 4800, 5900, 2000,5900, 6000]\n\n\ndef generate_large_list(ls, size):\n    n_copies = size // len(ls)\n    excess = size % len(ls)\n\n    result = sorted([element \n                     for i in range(n_copies) \n                     for element in ls] + ls[:excess]) \n    print(len(result))\n    return result\n\ndf = train.copy()\ndf['labels'] = [[ int(l) for l in x.split()] for x in df.Target.tolist()]\nminority_files = []\ntargets = []\nfor label in lb_to_balanced:\n    files = []\n    for i in range(len(df)):\n        if str(label) in df.iloc[i,1]:\n            files.append( df.iloc[i,0])\n    size = lb_to_added[lb_to_balanced.index(label)]\n    minority_files.extend( generate_large_list(files, size) )\n    targets.extend([str(label) for _ in range(size)])\n#     targets.extend(np.array([np.array([label]) for _ in range(size)]))\n\n\n# targets = np.array(targets)\nminority_dataframe = pd.DataFrame({'Id': minority_files, 'Target': targets})\nminority_dataframe['aug'] = 1\n\ntrain = train[['Id', 'Target', 'aug']]\ntrain = pd.concat([train, minority_dataframe], ignore_index=True)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-31T18:29:00.73321Z","iopub.execute_input":"2022-05-31T18:29:00.734136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def augment(image):\n    augment_img = iaa.Sequential([\n        iaa.OneOf([\n            iaa.Affine(rotate=0),\n            iaa.Affine(rotate=90),\n            iaa.Affine(rotate=180),\n            iaa.Affine(rotate=270),\n            iaa.Fliplr(0.5),\n            iaa.Flipud(0.5),\n        ])], random_order=True)\n\n    image_aug = augment_img.augment_image(image)\n    return image_aug","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input: list of image filters as png\n# Output: list of image filters as np.arrays\ndef image_to_arrays(path, aug):\n    \n    image_arrays = list()\n    for image in path:\n        array = np.asarray(Image.open(image))\n        if aug == 1:\n            array = augment(array)\n        image_arrays.append(array)\n        \n        \n    return image_arrays","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_blended_image(images): \n    # get rgby images for sample\n\n    # blend rgby images into single array\n    blended_array = np.stack(images[:-1], 2)\n\n    # Create PIL Image\n    blended_image = Image.fromarray( np.uint8(blended_array) )\n    return blended_image","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colours = ['_red.png', '_blue.png', '_yellow.png', '_green.png']\nTRAIN = '../input/human-protein-atlas-image-classification/train'\npaths = [[os.path.join(TRAIN, train.iloc[idx,0])+ colour for colour in colours] for idx in range(len(train))]\naugs = train.aug.values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUC_MODEL = \"./nuclei-model.pth\"\nCELL_MODEL = \"./cell-model.pth\"\nsegmentator = cellsegmentator.CellSegmentator(NUC_MODEL,CELL_MODEL,scale_factor=0.25,device=\"cuda\",padding=True,multi_channel_model=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_counts = []\nlabels = []\nfor label in train['Target']:\n    sep = label.split(' ')\n    for num in sep:\n        labels.append(int(num))\ncounts = pd.value_counts(labels)\n\n# It's an ugly plot, but I'm trying to save some time here...\nplt.bar(x = counts.index,height=counts)\nplt.xticks(counts.index)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = paths[3]\naug = augs[3]\narrays = image_to_arrays(image, aug)\nnuclei = arrays[1]\ncell = arrays[:-1]\n# Nuclei segmentation\nnuc_segmentations = segmentator.pred_nuclei([nuclei])\nf, ax = plt.subplots(1, 2, figsize=(16,16))\nax[0].imshow(arrays[1])\nax[0].set_title('Original Nucleis', size=20)\nax[1].imshow(nuc_segmentations[0])\nax[1].set_title('Segmented Nucleis', size=20)\nplt.show()\n#\n\n# Nuclei segmentation\nnuc_segmentations = segmentator.pred_nuclei([nuclei])\n    \n# Cell segmentation\ninter_step = [[i] for i in image[:-1]]\ncell_segmentations = segmentator.pred_cells(inter_step)\n\nf, ax = plt.subplots(1, 2, figsize=(16,16))\nax[0].imshow(get_blended_image(arrays))\nax[0].set_title('Original Cells', size=20)\nax[1].imshow(cell_segmentations[0])\nax[1].set_title('Segmented Cells', size=20)\nplt.show()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def iou(target, prediction):\n    intersection = np.logical_and(target, prediction)\n    union = np.logical_or(target, prediction)\n    iou_score_segmented_nuclies = np.sum(intersection) / np.sum(union)\n    return iou_score_segmented_nuclies","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\ndef generate_images(paths):\n    for i in range(len(paths)):\n        image = paths[i]\n        aug = augs[i]\n        arrays = image_to_arrays(image, aug)\n        yield arrays\n\n\niouNucleiSegmentationForDataset = 0\niouCellSegmentationForDataset = 0\nindex = 1\nimages = generate_images(paths)    \nimages_size = len(paths)\n\n\nfor arrays in images:\n    #print(aug)\n    #arrays = image_to_arrays(image, aug)\n    nuclei = arrays[1]\n    cell = arrays[:-1]\n    # Nuclei segmentation\n    nuc_segmentations = segmentator.pred_nuclei([nuclei])\n    nuc_segmented = cv2.cvtColor(nuc_segmentations[0], cv2.COLOR_BGR2GRAY)\n\n    # Cell segmentation\n    inter_step = [[i] for i in image[:-1]]\n    cell_segmentations = segmentator.pred_cells(inter_step)\n    cell_segmented = cv2.cvtColor(cell_segmentations[0], cv2.COLOR_BGR2GRAY)\n    \n    cell_nuclei_mask, cell_mask = label_cell(nuc_segmentations[0], cell_segmentations[0])    \n    \n#     original_cell = np.array(get_blended_image(arrays))\n    \n   # iouNucleiSegmentationForDataset += iou(nuc_segmented, nuclei_mask)\n    iouCellSegmentationForDataset += iou(cell_segmented, cell_mask)\n    print(f'{index} out of {len(paths)}')\n    index += 1\n    #break\nprint(f'segmented nuclies for the dataset is {iouNucleiSegmentationForDataset/images_size}')\nprint(f'Cell segmentation {iouCellSegmentationForDataset/images_size}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_nuclei_mask, cell_mask = label_cell(nuc_segmentations[0], cell_segmentations[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.imshow(get_blended_image(arrays))\nplt.imshow(cell_mask, alpha=0.7)\nplt.title('Segmentation results', size=40)\nplt.axis('off')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#original_cell = np.array(get_blended_image(arrays))\n#print(f'Segmented nuclies iou value is {iou(original_cell, nuc_segmentations[0])} for the image')\n#print(f'Cell segmentation iou value is {iou(original_cell, cell_segmentations[0])} for the image')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}