{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#CHANGE THIS TO LOCAL VALUES\n#set train.csv path\ncsv_path=\"../input/hpa-single-cell-image-classification/train.csv\"\n#set folder path of images in dataset\nimg_folder_path=\"../input/hpa-single-cell-image-classification/train/\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#OPTIONAL!\n#leave USE_PARTITIONS = False if the entire dataset should be used\n#use this to split up data if this is needed\n\nUSE_PARTITIONS = False #set to False to use entire dataset\nPARTS = 10 #amount of parts to split dataset up into\nCURRENT_PARTITION = 0 #if USE_PARTITIONS is True, this is the partition that is to be used in this run","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#DOES NOT HAVE TO BE CHANGED BUT CAN BE CHANGED\n#set model paths\n#if these files do not exist yet, the program will automatically download them\nNUC_MODEL = \"../input/hpacellsegmentatormodelweights/dpn_unet_nuclei_v1.pth\"\nCELL_MODEL = \"../input/hpacellsegmentatormodelweights/dpn_unet_cell_3ch_v1.pth\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#installs and imports\n!pip install https://github.com/CellProfiling/HPA-Cell-Segmentation/archive/master.zip\n!pip install bbox-visualizer\nimport pandas as pd\nimport numpy as np\nimport hpacellseg.cellsegmentator as cellsegmentator\nfrom hpacellseg.utils import label_cell, label_nuclei\nfrom PIL import Image, ImageDraw\nfrom tqdm import tqdm\nimport os.path\nimport matplotlib.pyplot as plt\nimport csv\nimport bbox_visualizer as bbv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#make a new directory for the masks\ndirName=\"masks\"\nif not os.path.exists(dirName):\n    os.mkdir(dirName)\n#read id/label csv to array\nid_labels_array=pd.read_csv(csv_path)\n#fix labels (convert to arrays)\nid_labels_array[\"Label\"]=id_labels_array[\"Label\"].apply(lambda x:list(map(int, x.split(\"|\"))))     \n#create list of all ids from the id_labels_array\nid_array=(id_labels_array[\"ID\"]).tolist()\n#create dictionary of all unique labels from the id_labels_array\nlabels=id_labels_array.set_index('ID').T.to_dict('list')\n#fix the dictionary format (2d arrays to 1D arrays)\nlabels = {num: labels[0] for num, labels in labels.items()}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def partition_data(ids,n_parts):\n    \"\"\"\n    input:\n        ids=list of image ids\n        n_parts=integer value of desired number of parts\n    output:\n        partition=dictionary of the ids that is split up into parts\n    \"\"\"\n    partition={}\n    parts = np.array_split(ids, n_parts)\n    for i,array in enumerate(parts):\n        partition[i]=array\n    return partition","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#CHOOSE THE PARTITION TO BE USED\nif USE_PARTITIONS==True:\n    id_array=partition_data(id_array,PARTS)\n    id_array=id_array[CURRENT_PARTITION]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segmentator = cellsegmentator.CellSegmentator(\n    NUC_MODEL,\n    CELL_MODEL,\n    scale_factor=0.25,\n    device=\"cuda\", #\"cuda\" for gpu, \"cpu\" for cpu\n    padding=False,\n    multi_channel_model=True,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#iterates through image ids and creates numpy arrays of the masks\nfor img_id in tqdm(id_array):\n    maskpath=\"masks/mask_\"+img_id+\".npy\"\n    if os.path.isfile(maskpath)==False:\n        path=img_folder_path+img_id\n        ch_r = Image.open(path+\"_red.png\")\n        ch_y = Image.open(path+\"_yellow.png\")\n        ch_b = Image.open(path+\"_blue.png\")\n        nuc_segmentations = segmentator.pred_nuclei([np.asarray(ch_b)])\n        cell_segmentations = segmentator.pred_cells([\n                [np.asarray(ch_r)],\n                [np.asarray(ch_y)],\n                [np.asarray(ch_b)]\n            ])\n        cell_nuclei_mask, cell_mask = label_cell(nuc_segmentations[0], cell_segmentations[0])\n        cell_mask = np.uint8(cell_mask)\n        np.save(maskpath,cell_mask)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize one of the masks\nimg_id=\"5c27f04c-bb99-11e8-b2b9-ac1f6b6435d0\" #random img id of img that has been converted\nmask_path=\"./masks/mask_\"+img_id+\".npy\"\nimg_r=np.asarray(Image.open(img_folder_path+img_id+\"_red.png\"))\nimg_g=np.asarray(Image.open(img_folder_path+img_id+\"_green.png\"))\nimg_b=np.asarray(Image.open(img_folder_path+img_id+\"_blue.png\"))\n\nplt.imshow(np.load(mask_path))\nplt.show()\nplt.imshow(np.dstack((img_r,img_g,img_b)))\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bbox(img,cell_id):\n    a = np.where(img == cell_id)\n    xmin, ymin, xmax, ymax = np.min(a[1]), np.min(a[0]), np.max(a[1]), np.max(a[0])\n    return xmin, ymin, xmax, ymax","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(r'bboxes.csv', 'a', newline='') as csvfile:\n    fieldnames = ['img_id','bboxes']\n    writer = csv.DictWriter(csvfile, fieldnames=fieldnames)\n    for img_id in tqdm(id_array):\n        cell_mask = np.load(\"./masks/mask_\"+img_id+'.npy')\n        cell_mask_flattened=np.ravel(cell_mask)\n        cell_ids=set(cell_mask_flattened)\n        cell_ids.remove(0)\n        bbox_dims=list()\n        for cell_id in cell_ids:\n            xmin, ymin, xmax, ymax = bbox(cell_mask,cell_id)\n            bbox_dims.append([xmin, ymin, xmax, ymax])\n        writer.writerow({'img_id':img_id, 'bboxes':bbox_dims})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize bbox results\nimg_id=\"5c27f04c-bb99-11e8-b2b9-ac1f6b6435d0\" #random img id of img that has been converted\nimg_r=np.asarray(Image.open(img_folder_path+img_id+\"_red.png\"))\nimg_g=np.asarray(Image.open(img_folder_path+img_id+\"_green.png\"))\nimg_b=np.asarray(Image.open(img_folder_path+img_id+\"_blue.png\"))\nimg=np.dstack((img_r,img_g,img_b))\ncell_mask = np.load(\"./masks/mask_\"+img_id+'.npy')\ncell_mask_flattened=np.ravel(cell_mask)\ncell_ids=set(cell_mask_flattened)\ncell_ids.remove(0)\nbboxes=list()\nfor cell_id in cell_ids:\n    xmin, ymin, xmax, ymax = bbox(cell_mask,cell_id)\n    bboxes.append([xmin, ymin, xmax, ymax])\nplt.imshow(bbv.draw_multiple_rectangles(img, bboxes))\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}