{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# install packages\n!rsync -a ../input/mmdetection-v280/mmdetection ../\n!pip install ../input/mmdetection-v280/src/mmdet-2.8.0/mmdet-2.8.0/\n!pip install ../input/mmdetection-v280/src/mmpycocotools-12.0.3/mmpycocotools-12.0.3/\n!pip install ../input/mmdetection-v280/src/addict-2.4.0-py3-none-any.whl\n!pip install ../input/mmdetection-v280/src/yapf-0.30.0-py2.py3-none-any.whl\n!pip install ../input/mmdetection-v280/src/mmcv_full-1.2.6-cp37-cp37m-manylinux1_x86_64.whl","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-05-23T12:17:44.570208Z","iopub.execute_input":"2021-05-23T12:17:44.570594Z","iopub.status.idle":"2021-05-23T12:18:31.700482Z","shell.execute_reply.started":"2021-05-23T12:17:44.570501Z","shell.execute_reply":"2021-05-23T12:18:31.699579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# basic \nimport pickle\nimport imageio\nimport warnings\nimport os, gc, cv2\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom itertools import groupby\n\nfrom tqdm import tqdm\nfrom multiprocessing import Pool\nimport base64\nimport typing as t\nimport zlib\nimport random\nrandom.seed(0)\n\n# visualize\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:21:12.049038Z","iopub.execute_input":"2021-05-23T12:21:12.04941Z","iopub.status.idle":"2021-05-23T12:21:12.925603Z","shell.execute_reply.started":"2021-05-23T12:21:12.049377Z","shell.execute_reply":"2021-05-23T12:21:12.924825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exp_name = \"v3\"\nconf_name = \"mask_rcnn_s101_fpn_syncbn-backbone+head_mstrain_1x_coco\"\n# mask_rcnn_s50_fpn_syncbn-backbone+head_mstrain_1x_coco.py\ncell_mask_dir = '../input/hpa-mask/hpa_cell_mask'    \nROOT = '../input/hpa-single-cell-image-classification/'\ntrain_or_test = 'train'\n\nimg_dir = f'../work/mmdet_{exp_name}_{train_or_test}'\nif not os.path.exists(img_dir):\n    os.makedirs(img_dir)\n#     !mkdir -p {img_dir}\n    \ndf = pd.read_csv(os.path.join(ROOT, 'train.csv'))\n\n# this script takes more than 9hours for full data.\n# debug = True\n# if debug:\n#     df = df[:4]","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:22.460893Z","iopub.execute_input":"2021-05-23T12:22:22.461223Z","iopub.status.idle":"2021-05-23T12:22:22.503148Z","shell.execute_reply.started":"2021-05-23T12:22:22.461195Z","shell.execute_reply":"2021-05-23T12:22:22.502269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Output directorys if we want to predict the masks\n# NUCL_DIR = '/kaggle/working/hpa-mask/hpa-nucl-mask'\n# CELL_DIR = '/kaggle/working/hpa-mask/hpa-cell-mask'\n# if not os.path.exists(NUCL_DIR):\n#     os.makedirs(NUCL_DIR)\n# if not os.path.exists(CELL_DIR):\n#     os.makedirs(CELL_DIR)\nPRE_LOADED_NUCL_DIR = '../input/hpa-mask/hpa_nuclei_mask'\nPRE_LOADED_CELL_DIR = '../input/hpa-mask/hpa_cell_mask'","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:24.608443Z","iopub.execute_input":"2021-05-23T12:22:24.608808Z","iopub.status.idle":"2021-05-23T12:22:24.615429Z","shell.execute_reply.started":"2021-05-23T12:22:24.608773Z","shell.execute_reply":"2021-05-23T12:22:24.614479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(ROOT)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:26.46692Z","iopub.execute_input":"2021-05-23T12:22:26.467242Z","iopub.status.idle":"2021-05-23T12:22:26.476783Z","shell.execute_reply.started":"2021-05-23T12:22:26.467213Z","shell.execute_reply":"2021-05-23T12:22:26.475913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(img_dir)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:33.103786Z","iopub.execute_input":"2021-05-23T12:22:33.104153Z","iopub.status.idle":"2021-05-23T12:22:33.109933Z","shell.execute_reply.started":"2021-05-23T12:22:33.104122Z","shell.execute_reply":"2021-05-23T12:22:33.109065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data exploration","metadata":{}},{"cell_type":"code","source":"df_train  = pd.read_csv(os.path.join(ROOT, train_or_test + '.csv'))\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:35.942528Z","iopub.execute_input":"2021-05-23T12:22:35.942884Z","iopub.status.idle":"2021-05-23T12:22:35.982616Z","shell.execute_reply.started":"2021-05-23T12:22:35.942842Z","shell.execute_reply":"2021-05-23T12:22:35.981711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'We have {df_train.shape[0]} rows and {df_train.shape[1]} columns in our df_train.csv.')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:37.946925Z","iopub.execute_input":"2021-05-23T12:22:37.947381Z","iopub.status.idle":"2021-05-23T12:22:37.955313Z","shell.execute_reply.started":"2021-05-23T12:22:37.947337Z","shell.execute_reply":"2021-05-23T12:22:37.954248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Missing values in train_df.csv in each columns:\\n{df_train.isnull().sum()}')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:41.909166Z","iopub.execute_input":"2021-05-23T12:22:41.909484Z","iopub.status.idle":"2021-05-23T12:22:41.918536Z","shell.execute_reply.started":"2021-05-23T12:22:41.909453Z","shell.execute_reply":"2021-05-23T12:22:41.917679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_labels = df_train.Label.unique().tolist()\nall_labels = '|'.join(all_labels)\nall_labels = all_labels.split('|')\nall_labels = list(set(all_labels))\nnum_unique_labels = len(all_labels)\nall_labels = sorted(all_labels, key=int)\nall_labels = ' '.join(all_labels)\nprint(f'{num_unique_labels} unique labels, values: {all_labels}')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:46.224909Z","iopub.execute_input":"2021-05-23T12:22:46.22523Z","iopub.status.idle":"2021-05-23T12:22:46.234666Z","shell.execute_reply.started":"2021-05-23T12:22:46.225202Z","shell.execute_reply":"2021-05-23T12:22:46.233681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['num_classes'] = df_train['Label'].apply(lambda r: len(r.split('|')))\ndf_train['num_classes'].value_counts().plot.bar(title='Examples with multiple labels', xlabel='number of labels per example', ylabel='# train examples')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:48.961082Z","iopub.execute_input":"2021-05-23T12:22:48.961478Z","iopub.status.idle":"2021-05-23T12:22:49.228535Z","shell.execute_reply.started":"2021-05-23T12:22:48.961443Z","shell.execute_reply":"2021-05-23T12:22:49.22758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = [str(i) for i in range(19)]\n\nunique_counts = {}\nfor lbl in labels:\n    unique_counts[lbl] = len(df_train[df_train.Label == lbl])\n# unique_counts\nfull_counts = {}\nfor lbl in labels:\n    count = 0\n    for row_label in df_train['Label']:\n        if lbl in row_label.split('|'): \n            count += 1\n    full_counts[lbl] = count\n  \ncounts = list(zip(full_counts.keys(), full_counts.values(), unique_counts.values()))\ncounts = np.array(sorted(counts, key=lambda x:-x[1]))\ncounts = pd.DataFrame(counts, columns=['label', 'full_count', 'unique_count'])\ncounts = counts.astype({\"label\": int, \"full_count\": int, \"unique_count\": int})\n                         \n# print (counts.dtypes)\ntype(counts[\"full_count\"][0])\nsns.set(style=\"whitegrid\")\nf, ax = plt.subplots(figsize=(16, 12))\n\nsns.set_color_codes(\"pastel\")\nsns.barplot(x=\"full_count\", y=\"label\", data=counts, order=counts.label.values,\n            label=\"full count\", color=\"b\", orient = 'h')\n\n# Plot the crashes where alcohol was involved\nsns.set_color_codes(\"muted\")\nsns.barplot(x=\"unique_count\", y=\"label\", data=counts, order=counts.label.values,\n            label=\"unique_count\", color=\"b\", orient = 'h')\n\n# Add a legend and informative axis label\nax.legend(ncol=2, loc=\"lower right\", frameon=True)\nax.set(ylabel=\"\",\n       xlabel=\"Counts\")\nsns.despine(left=True, bottom=True)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:22:51.229856Z","iopub.execute_input":"2021-05-23T12:22:51.230258Z","iopub.status.idle":"2021-05-23T12:22:52.171637Z","shell.execute_reply.started":"2021-05-23T12:22:51.230211Z","shell.execute_reply":"2021-05-23T12:22:52.170838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# helper funcs","metadata":{}},{"cell_type":"code","source":"def build_image_names(image_id: str, folder:str= train_or_test) -> list:\n    # mt is the mitchondria\n    mt = os.path.join(ROOT, folder, image_id + '_red.png')\n    \n    # er is the endoplasmic reticulum\n    er = os.path.join(ROOT, folder, image_id + '_yellow.png')\n    \n    # nu is the nuclei\n    nu = os.path.join(ROOT, folder, image_id + '_blue.png')\n    \n    return [mt], [er], [nu], [[mt], [er], [nu]]\n\ndef segmentCell(image, segmentator):\n    # For nuclei\n    nuc_segmentations = segmentator.pred_nuclei(image[2])\n    \n    # For full cells\n    cell_segmentations = segmentator.pred_cells(image)\n    \n    # post-processing\n    nuclei_mask, cell_mask = label_cell(nuc_segmentations[0], cell_segmentations[0])\n    \n    gc.collect(); del nuc_segmentations; del cell_segmentations\n    \n    return nuclei_mask, cell_mask \n\ndef plot_cell_segments(cell_mask, mt, er, nu):\n    \n    i = 0\n    microtubule = plt.imread(mt[i])    \n    endoplasmicrec = plt.imread(er[i])    \n    nuclei = plt.imread(nu[i])\n    img = np.dstack((microtubule, endoplasmicrec, nuclei))\n    \n    plt.figure(figsize=(15, 15))\n    plt.subplot(1, 3, 1)\n    plt.imshow(img)\n    plt.title('Image')\n    plt.axis('off')\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(cell_mask)\n    plt.title('Mask')\n    plt.axis('off')\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(img)\n    plt.imshow(cell_mask, alpha=0.6)\n    plt.title('Image + Mask')\n    plt.axis('off')\n    plt.show()\n    \n\n# convert segmentation mask image to run length encoding\nMAX_GREEN = 64 # filter out dark green cells\ndef get_rles_from_mask(image_id, class_id, image_size=None):\n    mask = np.load(f'{cell_mask_dir}/{image_id}.npz')['arr_0']\n    mask = cv2.resize(mask, dsize=image_size, interpolation=cv2.INTER_LINEAR)\n\n    if class_id != '18':\n        green_img = read_img(image_id, 'green', 'train', image_size)\n    rle_list = []\n    mask_ids = np.unique(mask)\n    for val in mask_ids:\n        if val == 0:\n            continue\n        binary_mask = np.where(mask == val, 1, 0).astype(bool)\n        if class_id != '18':\n            masked_img = green_img * binary_mask\n            #print(val, green_img.max(),masked_img.max())\n            if masked_img.max() < MAX_GREEN:\n                continue\n        rle = coco_rle_encode(binary_mask)\n        rle_list.append(rle)\n    return rle_list, mask.shape[0], mask.shape[1]\n\ndef coco_rle_encode(mask):\n    rle = {'counts': [], 'size': list(mask.shape)}\n    counts = rle.get('counts')\n    for i, (value, elements) in enumerate(groupby(mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle\n\n# mmdet custom dataset generator\ndef mk_mmdet_custom_data(image_id, class_id, image_size=None):\n    rles, height, width = get_rles_from_mask(image_id, class_id, image_size)\n    if len(rles) == 0:\n        return {\n            'filename': image_id+'.jpg',\n            'width': width,\n            'height': height,\n            'ann': {}\n        }\n    rles = mutils.frPyObjects(rles, height, width)\n    bboxes = mutils.toBbox(rles)\n    bboxes[:, 2] += bboxes[:, 0]\n    bboxes[:, 3] += bboxes[:, 1]\n    return {\n        'filename': image_id+'.jpg',\n        'width': width,\n        'height': height,\n        'ann':\n            {\n                'bboxes': np.array(bboxes, dtype=np.float32),\n                'labels': np.zeros(len(bboxes)), # dummy data.(will be replaced later)\n                'masks': rles\n            }\n    }\n\n# print utility from public notebook\ndef print_masked_img(image_id, mask):\n    img = load_RGBY_image(image_id, train_or_test)\n    \n    plt.figure(figsize=(15, 15))\n    plt.subplot(1, 3, 1)\n    plt.imshow(img)\n    plt.title('Image')\n    plt.axis('off')\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(mask)\n    plt.title('Mask')\n    plt.axis('off')\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(img)\n    plt.imshow(mask, alpha=0.6)\n    plt.title('Image + Mask')\n    plt.axis('off')\n    plt.show()\n    \n# image loader, using rgb only here\ndef load_RGBY_image(image_id, train_or_test='train', image_size=None):\n    red = read_img(image_id, \"red\", train_or_test, image_size)\n    green = read_img(image_id, \"green\", train_or_test, image_size)\n    blue = read_img(image_id, \"blue\", train_or_test, image_size)\n    #yellow = read_img(image_id, \"yellow\", train_or_test, image_size)\n    stacked_images = np.transpose(np.array([red, green, blue]), (1,2,0))\n    return stacked_images\n\n# \ndef read_img(image_id, color, train_or_test='train', image_size=None):\n    filename = f'{ROOT}/{train_or_test}/{image_id}_{color}.png'\n    assert os.path.exists(filename), f'not found {filename}'\n    img = cv2.imread(filename, cv2.IMREAD_UNCHANGED)\n    if image_size is not None:\n        img = cv2.resize(img, image_size)\n    if img.max() > 255:\n        img_max = img.max()\n        img = (img/255).astype('uint8')\n    return img\n\n# make annotation helper called multi processes\ndef mk_ann(idx):\n    image_id = df.iloc[idx].ID\n    class_id = df.iloc[idx].Label\n    image_size = (512,512)\n    anno = mk_mmdet_custom_data(image_id, class_id, image_size)\n    img = load_RGBY_image(image_id, train_or_test, image_size)\n    cv2.imwrite(f'{img_dir}/{image_id}.jpg', img)\n    return anno, idx, image_id","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:25:06.482744Z","iopub.execute_input":"2021-05-23T12:25:06.483122Z","iopub.status.idle":"2021-05-23T12:25:06.512655Z","shell.execute_reply.started":"2021-05-23T12:25:06.483092Z","shell.execute_reply":"2021-05-23T12:25:06.511616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# red = read_img('5c27f04c-bb99-11e8-b2b9-ac1f6b6435d0', \"red\", train_or_test)\n# red.dtype","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.527507Z","iopub.execute_input":"2021-05-23T00:45:40.527848Z","iopub.status.idle":"2021-05-23T00:45:40.539783Z","shell.execute_reply.started":"2021-05-23T00:45:40.527811Z","shell.execute_reply":"2021-05-23T00:45:40.539113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img = cv2.imread('../input/hpa-single-cell-image-classification/test/0173029a-161d-40ef-af28-2342915b22fb_blue.png', cv2.IMREAD_UNCHANGED)\n# img.dtype\n# print(img.max()/255)\n# if img.max() > 255:\n#     img_max = img.max()\n#     img = (img/255).astype('uint8')\n# img.dtype","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.541378Z","iopub.execute_input":"2021-05-23T00:45:40.541779Z","iopub.status.idle":"2021-05-23T00:45:40.549615Z","shell.execute_reply.started":"2021-05-23T00:45:40.541743Z","shell.execute_reply":"2021-05-23T00:45:40.548874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# checking segment mask\nTo extract the each cells, [CellSegmentator](https://github.com/CellProfiling/HPA-Cell-Segmentation) can be used.\nAnd The extracted segment masks are stored in [this dataset](https://www.kaggle.com/its7171/hpa-mask).\n","metadata":{}},{"cell_type":"markdown","source":"# Keeping only single label cells mask","metadata":{}},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:23:01.715903Z","iopub.execute_input":"2021-05-23T12:23:01.71626Z","iopub.status.idle":"2021-05-23T12:23:01.728971Z","shell.execute_reply.started":"2021-05-23T12:23:01.716229Z","shell.execute_reply":"2021-05-23T12:23:01.727904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"Label\"] = df_train[\"Label\"].str.split(\"|\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:23:04.05952Z","iopub.execute_input":"2021-05-23T12:23:04.059847Z","iopub.status.idle":"2021-05-23T12:23:04.088548Z","shell.execute_reply.started":"2021-05-23T12:23:04.059818Z","shell.execute_reply":"2021-05-23T12:23:04.087627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_train.loc[df_train['Label'].apply(lambda x: len(x)==1)==True]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T13:11:03.244261Z","iopub.execute_input":"2021-05-23T13:11:03.244605Z","iopub.status.idle":"2021-05-23T13:11:03.272865Z","shell.execute_reply.started":"2021-05-23T13:11:03.244575Z","shell.execute_reply":"2021-05-23T13:11:03.271668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-23T13:11:14.742818Z","iopub.execute_input":"2021-05-23T13:11:14.743207Z","iopub.status.idle":"2021-05-23T13:11:14.748522Z","shell.execute_reply.started":"2021-05-23T13:11:14.743176Z","shell.execute_reply":"2021-05-23T13:11:14.747324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting sgmentation masks for cell and nuclei","metadata":{}},{"cell_type":"code","source":"# !pip install -q \"../input/hpapytorchzoozip/pytorch_zoo-master\"\n# !pip install -q \"../input/hpacellsegmentatormaster/HPA-Cell-Segmentation-master\"","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.682293Z","iopub.execute_input":"2021-05-23T00:45:40.682739Z","iopub.status.idle":"2021-05-23T00:45:40.688354Z","shell.execute_reply.started":"2021-05-23T00:45:40.682704Z","shell.execute_reply":"2021-05-23T00:45:40.687228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import hpacellseg.cellsegmentator as cellsegmentator\n# from hpacellseg.utils import label_cell, label_nuclei\n\n# NUC_MODEL = '../input/hpacellsegmentatormodelweights/dpn_unet_nuclei_v1.pth'\n# CELL_MODEL = '../input/hpacellsegmentatormodelweights/dpn_unet_cell_3ch_v1.pth'\n\n# segmentator = cellsegmentator.CellSegmentator(\n#     NUC_MODEL,\n#     CELL_MODEL,\n#     scale_factor=0.25,\n#     device='cuda',\n#     padding=False,\n#     multi_channel_model=True\n# )","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.689973Z","iopub.execute_input":"2021-05-23T00:45:40.69084Z","iopub.status.idle":"2021-05-23T00:45:40.696706Z","shell.execute_reply.started":"2021-05-23T00:45:40.690801Z","shell.execute_reply":"2021-05-23T00:45:40.695618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the masks and saving them\n\n# for image_id in tqdm(train.ID.values):\n#     print('[INFO]: Dealing with {} ...'.format(image_id))\n#     mt, er, nu, images = build_image_names(image_id)    \n#     nucl_mask, cell_mask = segmentCell(images, segmentator)\n#     # Saving the predicted nucl and cell masks \n#     np.savez_compressed(f'{CELL_DIR}/{image_id}', cell_mask)\n#     np.savez_compressed(f'{NUCL_DIR}/{image_id}', nucl_mask)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.698548Z","iopub.execute_input":"2021-05-23T00:45:40.699119Z","iopub.status.idle":"2021-05-23T00:45:40.708293Z","shell.execute_reply.started":"2021-05-23T00:45:40.699081Z","shell.execute_reply":"2021-05-23T00:45:40.707591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading segmentation mask","metadata":{}},{"cell_type":"code","source":"# # os.listdir(ROOT+ train_or_test)\n# resized_cell_mask_dir = f'../work/mmdet_{exp_name}_cell_mask_resized'\n# !mkdir -p {resized_cell_mask_dir}\n# resized_nucl_mask_dir = f'../work/mmdet_{exp_name}_nucl_mask_resized'\n# !mkdir -p {resized_nucl_mask_dir}","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.710812Z","iopub.execute_input":"2021-05-23T00:45:40.711086Z","iopub.status.idle":"2021-05-23T00:45:40.71725Z","shell.execute_reply.started":"2021-05-23T00:45:40.711061Z","shell.execute_reply":"2021-05-23T00:45:40.716413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('../work')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.71899Z","iopub.execute_input":"2021-05-23T00:45:40.719653Z","iopub.status.idle":"2021-05-23T00:45:40.728272Z","shell.execute_reply.started":"2021-05-23T00:45:40.719616Z","shell.execute_reply":"2021-05-23T00:45:40.727612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = np.load(f'{cell_mask_dir}/5e9afd56-bb99-11e8-b2b9-ac1f6b6435d0.npz')['arr_0']\nplt.imshow(mask)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:40.730114Z","iopub.execute_input":"2021-05-23T00:45:40.730411Z","iopub.status.idle":"2021-05-23T00:45:41.145226Z","shell.execute_reply.started":"2021-05-23T00:45:40.730368Z","shell.execute_reply":"2021-05-23T00:45:41.144325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_cv2 = cv2.resize(mask, dsize=(512,512), interpolation=cv2.INTER_LINEAR)\nimg_cv2.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:41.146638Z","iopub.execute_input":"2021-05-23T00:45:41.14699Z","iopub.status.idle":"2021-05-23T00:45:41.167197Z","shell.execute_reply.started":"2021-05-23T00:45:41.146942Z","shell.execute_reply":"2021-05-23T00:45:41.166166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = '5e9afd56-bb99-11e8-b2b9-ac1f6b6435d0'\ncell_mask = np.load(f'{PRE_LOADED_CELL_DIR}/{image_id}.npz')['arr_0']\nnucl_mask = np.load(f'{PRE_LOADED_NUCL_DIR}/{image_id}.npz')['arr_0']\nmt, er, nu, images = build_image_names(image_id)    \nplot_cell_segments(cell_mask, mt, er, nu)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:41.168645Z","iopub.execute_input":"2021-05-23T00:45:41.168997Z","iopub.status.idle":"2021-05-23T00:45:43.498078Z","shell.execute_reply.started":"2021-05-23T00:45:41.16896Z","shell.execute_reply":"2021-05-23T00:45:43.497212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_mask_dir = '../input/hpa-mask/hpa_cell_mask'    \nfor idx in range(3):\n    image_id = df.iloc[idx].ID\n    cell_mask = np.load(f'{cell_mask_dir}/{image_id}.npz')['arr_0']\n    print_masked_img(image_id, cell_mask)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:43.499209Z","iopub.execute_input":"2021-05-23T00:45:43.499569Z","iopub.status.idle":"2021-05-23T00:45:51.467264Z","shell.execute_reply.started":"2021-05-23T00:45:43.499528Z","shell.execute_reply":"2021-05-23T00:45:51.466355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 0\nplt.figure(figsize=(15, 15))\nmicrotubule = plt.imread(mt[i])    \nendoplasmicrec = plt.imread(er[i])    \nnuclei = plt.imread(nu[i])\nmask = cell_mask\nimg = np.dstack((microtubule, endoplasmicrec, nuclei))\n# plt.imshow(img)\nplt.imshow(mask)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:51.468762Z","iopub.execute_input":"2021-05-23T00:45:51.469321Z","iopub.status.idle":"2021-05-23T00:45:52.431906Z","shell.execute_reply.started":"2021-05-23T00:45:51.469281Z","shell.execute_reply":"2021-05-23T00:45:52.431194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-23T00:45:52.433503Z","iopub.execute_input":"2021-05-23T00:45:52.43383Z","iopub.status.idle":"2021-05-23T00:45:52.439433Z","shell.execute_reply.started":"2021-05-23T00:45:52.433794Z","shell.execute_reply":"2021-05-23T00:45:52.438571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# generate data for mmdetection training","metadata":{}},{"cell_type":"code","source":"# from pycocotools import mask as mutils\n\n# # this part would take several hours, depends on your CPU power.\n# MAX_THRE = 4 # set your avarable CPU count.\n# p = Pool(processes=MAX_THRE)\n# annos = []\n# len_df = len(df)\n# for anno, idx, image_id in p.imap(mk_ann, range(len_df)):\n#     if len(anno['ann']) > 0:\n#         annos.append(anno)\n#     print(f'{idx+1}/{len_df}, {image_id}')\n    \n# lbl_cnt_dict = df.set_index('ID').to_dict()['Label']\n# trn_annos = []\n# val_annos = []\n# val_len = int(len(annos)*0.01)\n# for idx in range(len(annos)):\n#     ann = annos[idx]\n#     filename = ann['filename'].replace('.jpg','').replace('.png','')\n#     label_ids = lbl_cnt_dict[filename]\n#     len_ann = len(ann['ann']['bboxes'])\n#     bboxes = ann['ann']['bboxes']\n#     masks = ann['ann']['masks']\n#     # asign image level labels to each cells\n#     for cnt, label_id in enumerate(label_ids.split('|')):\n#         label_id = int(label_id)\n#         if cnt == 0:\n#             ann['ann']['labels'] = np.full(len_ann, label_id)\n#         else:\n#             ann['ann']['bboxes'] = np.concatenate([ann['ann']['bboxes'],bboxes])\n#             ann['ann']['labels'] = np.concatenate([ann['ann']['labels'],np.full(len_ann, label_id)])\n#             ann['ann']['masks'] = ann['ann']['masks'] + masks    \n#     if idx < val_len:\n#         val_annos.append(ann)\n#     else:\n#         trn_annos.append(ann)\n    \n# with open(f'../work/mmdet_{exp_name}_full.pkl', 'wb') as f:\n#     pickle.dump(annos, f)\n# with open(f'../work/mmdet_{exp_name}_trn.pkl', 'wb') as f:\n#     pickle.dump(trn_annos, f)\n# with open(f'../work/mmdet_{exp_name}_val.pkl', 'wb') as f:\n#     pickle.dump(val_annos, f)","metadata":{"lines_to_next_cell":2,"_kg_hide-output":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2021-05-23T12:42:22.161373Z","iopub.execute_input":"2021-05-23T12:42:22.161725Z","iopub.status.idle":"2021-05-23T12:42:22.167452Z","shell.execute_reply.started":"2021-05-23T12:42:22.161693Z","shell.execute_reply":"2021-05-23T12:42:22.166308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(f'../work/mmdet_{exp_name}_full.pkl', 'wb') as f:\n    pickle.dump(annos, f)\nwith open(f'../work/mmdet_{exp_name}_trn.pkl', 'wb') as f:\n    pickle.dump(trn_annos, f)\nwith open(f'../work/mmdet_{exp_name}_val.pkl', 'wb') as f:\n    pickle.dump(val_annos, f)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:43:01.496479Z","iopub.execute_input":"2021-05-23T12:43:01.496824Z","iopub.status.idle":"2021-05-23T12:43:01.66258Z","shell.execute_reply.started":"2021-05-23T12:43:01.496793Z","shell.execute_reply":"2021-05-23T12:43:01.661814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### I have already created the COCO format labled data so we can load it any time","metadata":{}},{"cell_type":"code","source":"with open(f'../work/mmdet_{exp_name}_full.pkl', 'rb') as f:\n          full_annos = pickle.load(f)\nwith open(f'../work/mmdet_{exp_name}_trn.pkl', 'rb') as f:\n          trn_annos = pickle.load(f)\nwith open(f'../work/mmdet_{exp_name}_val.pkl', 'rb') as f:\n          val_annos = pickle.load(f)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:44:44.824062Z","iopub.execute_input":"2021-05-23T12:44:44.824393Z","iopub.status.idle":"2021-05-23T12:44:46.629504Z","shell.execute_reply.started":"2021-05-23T12:44:44.824362Z","shell.execute_reply":"2021-05-23T12:44:46.628727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(trn_annos)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T13:11:26.81464Z","iopub.execute_input":"2021-05-23T13:11:26.814991Z","iopub.status.idle":"2021-05-23T13:11:26.822352Z","shell.execute_reply.started":"2021-05-23T13:11:26.81496Z","shell.execute_reply":"2021-05-23T13:11:26.821259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !cp '../input/loading-masks-creating-bboxs/mmdet_{exp_name}_full.pkl' ../work\n!cp '../input/loading-masks-creating-bboxs/mmdet_{exp_name}_trn.pkl' ../work\n!cp '../input/loading-masks-creating-bboxs/mmdet_{exp_name}_val.pkl' ../work","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:44:51.222413Z","iopub.execute_input":"2021-05-23T12:44:51.222762Z","iopub.status.idle":"2021-05-23T12:44:52.802065Z","shell.execute_reply.started":"2021-05-23T12:44:51.222732Z","shell.execute_reply":"2021-05-23T12:44:52.800942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2021-05-23T09:34:57.270704Z","iopub.execute_input":"2021-05-23T09:34:57.271086Z","iopub.status.idle":"2021-05-23T09:34:57.285916Z","shell.execute_reply.started":"2021-05-23T09:34:57.271057Z","shell.execute_reply":"2021-05-23T09:34:57.284904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = (512,512)\nfor idx in range(len(df)):\n    image_id = df.iloc[idx].ID\n    img = load_RGBY_image(image_id, train_or_test, image_size)\n    cv2.imwrite(f'{img_dir}/{image_id}.jpg', img)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T13:11:49.181805Z","iopub.execute_input":"2021-05-23T13:11:49.182174Z","iopub.status.idle":"2021-05-23T13:46:55.86938Z","shell.execute_reply.started":"2021-05-23T13:11:49.182141Z","shell.execute_reply":"2021-05-23T13:46:55.868477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir(img_dir))","metadata":{"execution":{"iopub.status.busy":"2021-05-23T13:10:41.32698Z","iopub.execute_input":"2021-05-23T13:10:41.327296Z","iopub.status.idle":"2021-05-23T13:10:41.334418Z","shell.execute_reply.started":"2021-05-23T13:10:41.327267Z","shell.execute_reply":"2021-05-23T13:10:41.333466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# training","metadata":{}},{"cell_type":"code","source":"# I just made following config files based on default mask_rcnn.\n# The main changes are CustomDataset, num_classes, data path, etc.\n# Other than that, I used it as is for mmdetection.\n!ls -l ../mmdetection/configs/hpa_{exp_name}/","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:53:31.848242Z","iopub.execute_input":"2021-05-23T12:53:31.848603Z","iopub.status.idle":"2021-05-23T12:53:32.504277Z","shell.execute_reply.started":"2021-05-23T12:53:31.84857Z","shell.execute_reply":"2021-05-23T12:53:32.503146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# conf_name = \"mask_rcnn_s101_fpn_syncbn-backbone+head_mstrain_1x_coco\"\nconf_name = \"mask_rcnn_s50_fpn_syncbn-backbone+head_mstrain_1x_coco\"\nprint(conf_name)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:55:29.391276Z","iopub.execute_input":"2021-05-23T12:55:29.391607Z","iopub.status.idle":"2021-05-23T12:55:29.399449Z","shell.execute_reply.started":"2021-05-23T12:55:29.391576Z","shell.execute_reply":"2021-05-23T12:55:29.398504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = f'configs/hpa_{exp_name}/{conf_name}.py'\n\n# using --no-validate to avoid some errors for custom dataset metrics\nadditional_conf = '--no-validate --cfg-options'\nadditional_conf += f' work_dir=../working/work_dir'\nadditional_conf += f' optimizer.lr=0.0025'\ncmd = f'bash -x tools/dist_train.sh {config} 1 {additional_conf}'\n!cd ../mmdetection; {cmd}","metadata":{"execution":{"iopub.status.busy":"2021-05-23T12:55:31.14474Z","iopub.execute_input":"2021-05-23T12:55:31.145103Z","iopub.status.idle":"2021-05-23T12:56:13.765444Z","shell.execute_reply.started":"2021-05-23T12:55:31.14507Z","shell.execute_reply":"2021-05-23T12:56:13.764391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -Rl .","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:16:24.356049Z","iopub.status.idle":"2021-05-23T01:16:24.356746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}