{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2 as cv\n\nfrom PIL import Image, ImageEnhance\nimport matplotlib.pyplot as plt\n\nimport plotly.express as px\n\nimport json\nfrom collections import defaultdict","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-25T04:22:21.106198Z","iopub.execute_input":"2021-11-25T04:22:21.106626Z","iopub.status.idle":"2021-11-25T04:22:22.988901Z","shell.execute_reply.started":"2021-11-25T04:22:21.106576Z","shell.execute_reply":"2021-11-25T04:22:22.987865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pycocotools","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:22.990718Z","iopub.execute_input":"2021-11-25T04:22:22.99097Z","iopub.status.idle":"2021-11-25T04:22:44.003041Z","shell.execute_reply.started":"2021-11-25T04:22:22.990938Z","shell.execute_reply":"2021-11-25T04:22:44.001931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycocotools import _mask as MeaskUtils","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:44.004887Z","iopub.execute_input":"2021-11-25T04:22:44.005235Z","iopub.status.idle":"2021-11-25T04:22:44.015202Z","shell.execute_reply.started":"2021-11-25T04:22:44.005152Z","shell.execute_reply":"2021-11-25T04:22:44.014164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Refferences\n1. https://www.kaggle.com/dschettler8845/train-sartorius-segmentation-eda-effdet-tf\n2. liveCell dataset https://sartorius-research.github.io/LIVECell/","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/sartorius-cell-instance-segmentation/train | wc -l","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:44.018906Z","iopub.execute_input":"2021-11-25T04:22:44.019331Z","iopub.status.idle":"2021-11-25T04:22:44.858919Z","shell.execute_reply.started":"2021-11-25T04:22:44.019254Z","shell.execute_reply":"2021-11-25T04:22:44.858001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/sartorius-cell-instance-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:44.87111Z","iopub.execute_input":"2021-11-25T04:22:44.871717Z","iopub.status.idle":"2021-11-25T04:22:45.543344Z","shell.execute_reply.started":"2021-11-25T04:22:44.871658Z","shell.execute_reply":"2021-11-25T04:22:45.542053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:45.545785Z","iopub.execute_input":"2021-11-25T04:22:45.546209Z","iopub.status.idle":"2021-11-25T04:22:45.596762Z","shell.execute_reply.started":"2021-11-25T04:22:45.546158Z","shell.execute_reply":"2021-11-25T04:22:45.595285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n# modified from: https://www.kaggle.com/inversion/run-length-decoding-quick-start\ndef rle_decode(mask_rle, shape, color=1):\n    \"\"\" TBD\n    \n    Args:\n        mask_rle (str): run-length as string formated (start length)\n        shape (tuple of ints): (height,width) of array to return \n    \n    Returns: \n        Mask (np.array)\n            - 1 indicating mask\n            - 0 indicating background\n\n    \"\"\"\n    # Split the string by space, then convert it into a integer array\n    s = np.array(mask_rle.split(), dtype=int)\n\n    # Every even value is the start, every odd value is the \"run\" length\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n\n    # The image image is actually flattened since RLE is a 1D \"run\"\n    if len(shape)==3:\n        h, w, d = shape\n        img = np.zeros((h * w, d), dtype=np.float32)\n    else:\n        h, w = shape\n        img = np.zeros((h * w,), dtype=np.float32)\n\n    # The color here is actually just any integer you want!\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = color\n        \n    # Don't forget to change the image back to the original shape\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:45.601918Z","iopub.execute_input":"2021-11-25T04:22:45.602389Z","iopub.status.idle":"2021-11-25T04:22:45.615344Z","shell.execute_reply.started":"2021-11-25T04:22:45.602327Z","shell.execute_reply":"2021-11-25T04:22:45.614144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img_and_mask(img_path, annotation, width, height, mask_only=False, rle_fn=rle_decode):\n    \"\"\" Capture the relevant image array as well as the image mask \"\"\"\n    img_mask = np.zeros((height, width), dtype=np.uint8)\n    for i, annot in enumerate(annotation): \n        img_mask = np.where(rle_fn(annot, (height, width))!=0, i, img_mask)\n    \n    # Early Exit\n    if mask_only:\n        return img_mask\n    \n    # Else Return images\n    #img = tf_load_png(img_path)[..., 0]\n    img = cv.imread(img_path)\n    return img, img_mask\n\ndef plot_img_and_mask(img, mask, bboxes=None, invert_img=True, boost_contrast=True):\n    \"\"\" Function to take an image and the corresponding mask and plot\n    \n    Args:\n        img (np.arr): 1 channel np arr representing the image of cellular structures\n        mask (np.arr): 1 channel np arr representing the instance masks (incrementing by one)\n        bboxes (list of tuples, optional): (tl, br) coordinates of enclosing bboxes\n        invert_img (bool, optional): Whether or not to invert the base image\n        boost_contrast (bool, optional): Whether or not to boost contrast of the base image\n        \n    Returns:\n        None; Plots the two arrays and overlays them to create a merged image\n    \"\"\"\n    plt.figure(figsize=(20,10))\n    \n    plt.subplot(1,3,1)\n    #_img = np.tile(np.expand_dims(img, axis=-1), 3)\n    _img= img\n    \n    # Flip black-->white ... white-->black\n    if invert_img:\n        _img = _img.max()-_img\n    \n    if boost_contrast:\n        _img = np.asarray(ImageEnhance.Contrast(Image.fromarray(_img)).enhance(16))\n    \n    if bboxes:\n        for i, bbox in enumerate(bboxes):\n            mask = cv.rectangle(mask, bbox[0], bbox[1], (i+1, 0, 0), thickness=2)\n    \n    plt.imshow(_img)\n    plt.axis(False)\n    plt.title(\"Cell Image\", fontweight=\"bold\")\n    \n    plt.subplot(1,3,2)\n    _mask = np.zeros_like(_img)\n    _mask[..., 0] = mask\n    plt.imshow(mask, cmap=\"inferno\")\n    plt.axis(False)\n    plt.title(\"Instance Segmentation Mask\", fontweight=\"bold\")\n    \n    merged = cv.addWeighted(_img, 0.75, np.clip(_mask, 0, 1)*255, 0.25, 0.0,)\n    plt.subplot(1,3,3)\n    plt.imshow(merged)\n    plt.axis(False)\n    plt.title(\"Cell Image w/ Instance Segmentation Mask Overlay\", fontweight=\"bold\")\n    \n    plt.tight_layout()\n    plt.show()\n    \n# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    \"\"\" TBD\n    \n    Args:\n        img (np.array): \n            - 1 indicating mask\n            - 0 indicating background\n    \n    Returns: \n        run length as string formated\n    \"\"\"\n    \n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:45.616585Z","iopub.execute_input":"2021-11-25T04:22:45.617142Z","iopub.status.idle":"2021-11-25T04:22:45.644812Z","shell.execute_reply.started":"2021-11-25T04:22:45.617105Z","shell.execute_reply":"2021-11-25T04:22:45.643821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://github.com/PyImageSearch/imutils/blob/master/imutils/convenience.py\ndef grab_contours(cnts):\n    \"\"\" TBD \"\"\"\n    \n    # if the length the contours tuple returned by cv2.findContours\n    # is '2' then we are using either OpenCV v2.4, v4-beta, or\n    # v4-official\n    if len(cnts) == 2:\n        cnts = cnts[0]\n\n    # if the length of the contours tuple is '3' then we are using\n    # either OpenCV v3, v4-pre, or v4-alpha\n    elif len(cnts) == 3:\n        cnts = cnts[1]\n\n    # otherwise OpenCV has changed their cv2.findContours return\n    # signature yet again and I have no idea WTH is going on\n    else:\n        raise Exception((\"Contours tuple must have length 2 or 3, \"\n            \"otherwise OpenCV changed their cv2.findContours return \"\n            \"signature yet again. Refer to OpenCV's documentation \"\n            \"in that case\"))\n\n    # return the actual contours array\n    return cnts\n\ndef get_contour_bbox(msk):\n    \"\"\" Function to return the bounding box (tl, br) for a given mask \"\"\"\n    \n    # Get contour(s) --> There should be only one\n    cnts = cv.findContours(msk.copy(), cv.RETR_EXTERNAL, cv.CHAIN_APPROX_SIMPLE)\n\n    contour = grab_contours(cnts)\n    \n    if len(contour)==0:\n        return None\n    else:\n        contour = contour[0]\n    \n    # Get extreme coordinates\n    tl = (tuple(contour[contour[:, :, 0].argmin()][0])[0], \n          tuple(contour[contour[:, :, 1].argmin()][0])[1])\n    br = (tuple(contour[contour[:, :, 0].argmax()][0])[0], \n          tuple(contour[contour[:, :, 1].argmax()][0])[1])\n    return tl, br","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:45.64596Z","iopub.execute_input":"2021-11-25T04:22:45.647091Z","iopub.status.idle":"2021-11-25T04:22:45.677085Z","shell.execute_reply.started":"2021-11-25T04:22:45.647038Z","shell.execute_reply":"2021-11-25T04:22:45.675786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Train Images","metadata":{}},{"cell_type":"code","source":"TRAIN_DIR = '/kaggle/input/sartorius-cell-instance-segmentation/train'","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:45.678614Z","iopub.execute_input":"2021-11-25T04:22:45.679022Z","iopub.status.idle":"2021-11-25T04:22:45.702629Z","shell.execute_reply.started":"2021-11-25T04:22:45.678978Z","shell.execute_reply":"2021-11-25T04:22:45.70115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aggregate under training \ntrain_df[\"img_path\"] = train_df[\"id\"].apply(lambda x: os.path.join(TRAIN_DIR, x+\".png\")) # Capture Image Path As Well\ntmp_df = train_df.drop_duplicates(subset=[\"id\", \"img_path\"]).reset_index(drop=True)\ntmp_df[\"annotation\"] = train_df.groupby(\"id\")[\"annotation\"].agg(list).reset_index(drop=True)\ntrain_df = tmp_df.copy()\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:45.704239Z","iopub.execute_input":"2021-11-25T04:22:45.70455Z","iopub.status.idle":"2021-11-25T04:22:46.043988Z","shell.execute_reply.started":"2021-11-25T04:22:45.704516Z","shell.execute_reply":"2021-11-25T04:22:46.042896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for _, _dirname, files in os.walk(TRAIN_DIR):\n    nums = 1\n    for item in files[:nums]:\n        _filepath = os.path.join(TRAIN_DIR, item)\n        print(_filepath)\n        img = cv.imread(_filepath)\n        plt.figure()\n        plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:46.046485Z","iopub.execute_input":"2021-11-25T04:22:46.046897Z","iopub.status.idle":"2021-11-25T04:22:46.496734Z","shell.execute_reply.started":"2021-11-25T04:22:46.046846Z","shell.execute_reply":"2021-11-25T04:22:46.495734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Only show Single Mask","metadata":{}},{"cell_type":"code","source":"_idx = 18\n_one = train_df.iloc[_idx]\n_one","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:46.498386Z","iopub.execute_input":"2021-11-25T04:22:46.498848Z","iopub.status.idle":"2021-11-25T04:22:46.510368Z","shell.execute_reply.started":"2021-11-25T04:22:46.498793Z","shell.execute_reply":"2021-11-25T04:22:46.508636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_filepath = os.path.join(TRAIN_DIR, _one.id + \".png\")\n\n_img = cv.imread(_filepath)\n\n_img = np.asarray(ImageEnhance.Contrast(Image.fromarray(_img)).enhance(16))\n\n\nplt.figure()\nplt.imshow(_img)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:46.512276Z","iopub.execute_input":"2021-11-25T04:22:46.512695Z","iopub.status.idle":"2021-11-25T04:22:46.955335Z","shell.execute_reply.started":"2021-11-25T04:22:46.512649Z","shell.execute_reply":"2021-11-25T04:22:46.951984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getAllAnnotation(annotation, height, width, rle_fn):\n    img_mask = np.zeros((height, width), dtype=np.uint8)\n    for i, annot in enumerate(annotation):\n        img_mask = np.where(rle_fn(annot, (height, width))!=0, i, img_mask)\n    return img_mask ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:46.956935Z","iopub.execute_input":"2021-11-25T04:22:46.957209Z","iopub.status.idle":"2021-11-25T04:22:46.968277Z","shell.execute_reply.started":"2021-11-25T04:22:46.957176Z","shell.execute_reply":"2021-11-25T04:22:46.964613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#_mask = rle_decode(_one.annotation, (_one.height, _one.width))\n_mask = getAllAnnotation(_one.annotation, _one.height, _one.width, rle_decode)\nplt.figure()\nplt.imshow(_mask, cmap=\"inferno\")","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:46.970324Z","iopub.execute_input":"2021-11-25T04:22:46.971401Z","iopub.status.idle":"2021-11-25T04:22:47.37399Z","shell.execute_reply.started":"2021-11-25T04:22:46.971257Z","shell.execute_reply":"2021-11-25T04:22:47.373219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_mask2 = np.zeros_like(_img)\n_mask2[..., 0] = _mask\n_merged = cv.addWeighted(_img, 0.75, np.clip(_mask2, 0, 1)*255, 0.25, 0.0,)\nplt.figure()\nplt.imshow(_merged, cmap=\"inferno\")","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:47.375508Z","iopub.execute_input":"2021-11-25T04:22:47.375778Z","iopub.status.idle":"2021-11-25T04:22:47.78217Z","shell.execute_reply.started":"2021-11-25T04:22:47.375745Z","shell.execute_reply":"2021-11-25T04:22:47.781017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n... WIDTH VALUE COUNTS ...\")\nfor k,v in train_df.width.value_counts().items():\n    print(f\"\\t--> There are {v} images with WIDTH={k}\")\n\nprint(\"\\n\\n... HEIGHT VALUE COUNTS ...\")\nfor k,v in train_df.height.value_counts().items():\n    print(f\"\\t--> There are {v} images with HEIGHT={k}\")\n\nprint(\"\\n\\n... AREA COUNTS ...\")\nfor k,v in (train_df.width*train_df.height).value_counts().items():\n    print(f\"\\t--> There are {v} images with AREA={k}\")\n\nprint(\"\\n\\n... NOTE: ALL THE IMAGES ARE THE SAME SIZE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:47.783554Z","iopub.execute_input":"2021-11-25T04:22:47.783789Z","iopub.status.idle":"2021-11-25T04:22:47.799755Z","shell.execute_reply.started":"2021-11-25T04:22:47.783757Z","shell.execute_reply":"2021-11-25T04:22:47.798288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n... PLATE TIME VALUE COUNTS ...\")\nfor k,v in train_df.plate_time.value_counts().items():\n    print(f\"\\t--> There are {v} images with PLATE_TIME={k}\")\nfig = px.histogram(train_df, x=\"plate_time\", color=\"cell_type\", title=\"<b>Plate Time Histogram</b>\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:47.802128Z","iopub.execute_input":"2021-11-25T04:22:47.80306Z","iopub.status.idle":"2021-11-25T04:22:49.405751Z","shell.execute_reply.started":"2021-11-25T04:22:47.802984Z","shell.execute_reply":"2021-11-25T04:22:49.405015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n... SAMPLE DATE VALUE COUNTS ...\")\nfor k,v in train_df.sample_date.value_counts().items():\n    print(f\"\\t--> There are {v} images with SAMPLE_DATE={k}\")\nfig = px.histogram(train_df, train_df.sample_date.apply(lambda x: x.replace(\"-\", \"_\")), color=\"cell_type\", title=\"<b>Sample Date Value Histogram</b>\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:49.407405Z","iopub.execute_input":"2021-11-25T04:22:49.407854Z","iopub.status.idle":"2021-11-25T04:22:49.507119Z","shell.execute_reply.started":"2021-11-25T04:22:49.407802Z","shell.execute_reply":"2021-11-25T04:22:49.505594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n... ELAPSED TIME DELTA VALUE COUNTS ...\")\nfor k,v in train_df.elapsed_timedelta.value_counts().items():\n    print(f\"\\t--> There are {v} images with SAMPLE_DATE={k}\")\nfig = px.histogram(train_df, \"elapsed_timedelta\", color=\"cell_type\", title=\"<b>Elapsed Time Delta Value Histogram</b>\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:49.509286Z","iopub.execute_input":"2021-11-25T04:22:49.509599Z","iopub.status.idle":"2021-11-25T04:22:49.611535Z","shell.execute_reply.started":"2021-11-25T04:22:49.509566Z","shell.execute_reply":"2021-11-25T04:22:49.610136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n... SAMPLE ID VALUE COUNTS (>1) ...\")\nprint(f\"\\t--> There are {len(train_df[train_df.sample_id.isin([x for x,v in train_df.sample_id.value_counts().items() if v>1])])} SAMPLE_IDs with more than one image\\n\")\nfor k,v in train_df[train_df.sample_id.isin([x for x,v in train_df.sample_id.value_counts().items() if v>1])].reset_index()[\"sample_id\"].value_counts().items():\n    print(f\"\\t--> There are {v} images with SAMPLE_ID={k}\")\nfig = px.histogram(train_df[train_df.sample_id.isin([x for x,v in train_df.sample_id.value_counts().items() if v>1])].reset_index(), \"sample_id\", color=\"cell_type\", title=\"<b>Sample ID Value Histogram</b>\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:49.61295Z","iopub.execute_input":"2021-11-25T04:22:49.613308Z","iopub.status.idle":"2021-11-25T04:22:49.748833Z","shell.execute_reply.started":"2021-11-25T04:22:49.613262Z","shell.execute_reply":"2021-11-25T04:22:49.742916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n\\n... CELL TYPE VALUE COUNTS ...\")\nfor k,v in train_df.cell_type.value_counts().items():\n    print(f\"\\t--> There are {v} images with CELL_TYPE={k}\")\n    \nfig = px.histogram(train_df, x=\"cell_type\", title=\"<b>Cell Type Histogram</b>\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:49.750571Z","iopub.execute_input":"2021-11-25T04:22:49.751087Z","iopub.status.idle":"2021-11-25T04:22:49.84069Z","shell.execute_reply.started":"2021-11-25T04:22:49.750951Z","shell.execute_reply":"2021-11-25T04:22:49.839747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CELL_TYPES = list(train_df.cell_type.unique())\nfor ct in CELL_TYPES:\n    print(f\"\\n\\n... SHOWING THREE EXAMPLES OF CELL TYPE {ct.upper()} ...\\n\")\n    for i in range(3):\n        img, msk = get_img_and_mask(**train_df[train_df.cell_type==ct][[\"img_path\", \"annotation\", \"width\", \"height\"]].sample(3).reset_index(drop=True).iloc[i].to_dict())\n        plot_img_and_mask(img, msk)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:49.842321Z","iopub.execute_input":"2021-11-25T04:22:49.842568Z","iopub.status.idle":"2021-11-25T04:22:57.298222Z","shell.execute_reply.started":"2021-11-25T04:22:49.842538Z","shell.execute_reply":"2021-11-25T04:22:57.297146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show External Data","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/sartorius-cell-instance-segmentation\"\nLC_DIR = os.path.join(DATA_DIR, \"LIVECell_dataset_2021\")\nLC_ANN_DIR = os.path.join(LC_DIR, \"annotations\")\nLC_IMG_DIR = os.path.join(LC_DIR, \"images\")","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:57.303884Z","iopub.execute_input":"2021-11-25T04:22:57.304187Z","iopub.status.idle":"2021-11-25T04:22:57.310098Z","shell.execute_reply.started":"2021-11-25T04:22:57.304152Z","shell.execute_reply":"2021-11-25T04:22:57.308816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LC_CELL_TYPES = os.listdir(os.path.join(LC_ANN_DIR, \"LIVECell_single_cells\"))\n\nprint(\"\\n... LOADING TRAIN COCO JSON ...\\n\")\nLC_COCO_TRAIN = os.path.join(LC_ANN_DIR, \"LIVECell\", \"livecell_coco_train.json\")\n\nprint(\"\\n... LOADING VALIDATION COCO JSON ...\\n\")\nLC_COCO_VAL = os.path.join(LC_ANN_DIR, \"LIVECell\", \"livecell_coco_val.json\")\n\nprint(\"\\n... LOADING TEST COCO JSON ...\\n\")\nLC_COCO_TEST = os.path.join(LC_ANN_DIR, \"LIVECell\", \"livecell_coco_test.json\")\n\nLC_SC_TRAIN = {\n    lc_ct:os.path.join(LC_ANN_DIR, \"LIVECell_single_cells\", lc_ct, f\"livecell_{lc_ct}_train.json\") \\\n    for lc_ct in LC_CELL_TYPES\n}\nLC_SC_VAL = {\n    lc_ct:os.path.join(LC_ANN_DIR, \"LIVECell_single_cells\", lc_ct, f\"livecell_{lc_ct}_val.json\") \\\n    for lc_ct in LC_CELL_TYPES\n}\nLC_SC_TEST = {\n    lc_ct:os.path.join(LC_ANN_DIR, \"LIVECell_single_cells\", lc_ct, f\"livecell_{lc_ct}_test.json\") \\\n    for lc_ct in LC_CELL_TYPES\n}\n\nprint(LC_SC_TRAIN)\n#print(LC_SC_VAL)\n#print(LC_SC_TEST)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:57.311712Z","iopub.execute_input":"2021-11-25T04:22:57.312984Z","iopub.status.idle":"2021-11-25T04:22:57.348852Z","shell.execute_reply.started":"2021-11-25T04:22:57.31291Z","shell.execute_reply":"2021-11-25T04:22:57.347872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tifffile","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:57.350448Z","iopub.execute_input":"2021-11-25T04:22:57.350971Z","iopub.status.idle":"2021-11-25T04:22:57.527072Z","shell.execute_reply.started":"2021-11-25T04:22:57.350924Z","shell.execute_reply":"2021-11-25T04:22:57.52561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LiveCell\n\nhttps://www.nature.com/articles/s41592-021-01249-6","metadata":{}},{"cell_type":"code","source":"LC_DIR_IMGS = os.path.join(LC_DIR, 'images')\n#tiffiles = []\ntiffiles = defaultdict(list)\nclass_count = defaultdict(list)\n\nfor _dirs in os.listdir(LC_DIR_IMGS):\n    _split = os.path.join(LC_DIR_IMGS, _dirs)\n    class_count[_dirs] = {}\n    for foldlist in os.listdir(_split):\n        _cellfold = os.path.join(_split, foldlist)\n        \n        class_count[_dirs][foldlist] = 0\n        \n        for file in os.listdir(_cellfold):\n            _tiffilepath = os.path.join(_cellfold, file)\n            #tiffiles.append(_tiffilepath)\n            tiffiles[file] = _tiffilepath\n            class_count[_dirs][foldlist] +=1","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:29:23.138268Z","iopub.execute_input":"2021-11-25T04:29:23.140289Z","iopub.status.idle":"2021-11-25T04:29:23.194633Z","shell.execute_reply.started":"2021-11-25T04:29:23.14021Z","shell.execute_reply":"2021-11-25T04:29:23.193395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_count","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:29:24.77724Z","iopub.execute_input":"2021-11-25T04:29:24.77758Z","iopub.status.idle":"2021-11-25T04:29:24.786142Z","shell.execute_reply.started":"2021-11-25T04:29:24.777547Z","shell.execute_reply":"2021-11-25T04:29:24.785385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO convert tiff to img\nfilename = 'SHSY5Y_Phase_B10_2_03d08h00m_4.tif'\nf = tiffiles[filename]\nimage = tifffile.imread(f)\nplt.figure()\nplt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:57.897256Z","iopub.execute_input":"2021-11-25T04:22:57.897665Z","iopub.status.idle":"2021-11-25T04:22:58.291794Z","shell.execute_reply.started":"2021-11-25T04:22:57.897634Z","shell.execute_reply":"2021-11-25T04:22:58.290529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_live_cell_annotation_file = '../input/sartorius-cell-instance-segmentation/LIVECell_dataset_2021/annotations/LIVECell/livecell_coco_train.json'\n_live_cell_shsy5y_train = '../input/sartorius-cell-instance-segmentation/LIVECell_dataset_2021/annotations/LIVECell_single_cells/shsy5y/livecell_shsy5y_train.json'\n_live_cell_shsy5y_val = '../input/sartorius-cell-instance-segmentation/LIVECell_dataset_2021/annotations/LIVECell_single_cells/shsy5y/livecell_shsy5y_val.json'\n_live_cell_shsy5y_test = '../input/sartorius-cell-instance-segmentation/LIVECell_dataset_2021/annotations/LIVECell_single_cells/shsy5y/livecell_shsy5y_test.json'","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:32:39.678168Z","iopub.execute_input":"2021-11-25T04:32:39.678583Z","iopub.status.idle":"2021-11-25T04:32:39.68393Z","shell.execute_reply.started":"2021-11-25T04:32:39.678544Z","shell.execute_reply":"2021-11-25T04:32:39.682904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_dataset = json.loads(open(_live_cell_annotation_file).read())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:22:58.300943Z","iopub.execute_input":"2021-11-25T04:22:58.301238Z","iopub.status.idle":"2021-11-25T04:23:30.432482Z","shell.execute_reply.started":"2021-11-25T04:22:58.301205Z","shell.execute_reply":"2021-11-25T04:23:30.431487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_dataset_shsy5y_train = json.loads(open(_live_cell_shsy5y_train).read())\n_dataset_shsy5y_val = json.loads(open(_live_cell_shsy5y_val).read())\n_dataset_shsy5y_test = json.loads(open(_live_cell_shsy5y_test).read())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:32:52.145951Z","iopub.execute_input":"2021-11-25T04:32:52.146565Z","iopub.status.idle":"2021-11-25T04:33:02.291517Z","shell.execute_reply.started":"2021-11-25T04:32:52.146516Z","shell.execute_reply":"2021-11-25T04:33:02.290779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(_dataset['info'])\nprint(_dataset['licenses'])\nprint(_dataset['categories'])","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:24:28.532204Z","iopub.execute_input":"2021-11-25T04:24:28.532513Z","iopub.status.idle":"2021-11-25T04:24:28.537991Z","shell.execute_reply.started":"2021-11-25T04:24:28.532482Z","shell.execute_reply":"2021-11-25T04:24:28.536978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(_dataset.keys())\nprint(_dataset['images'][0].keys())\nprint(_dataset['annotations']['2'].keys())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:24:29.945235Z","iopub.execute_input":"2021-11-25T04:24:29.945539Z","iopub.status.idle":"2021-11-25T04:24:29.952025Z","shell.execute_reply.started":"2021-11-25T04:24:29.945508Z","shell.execute_reply":"2021-11-25T04:24:29.95104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(_dataset_shsy5y_train['images']))\nprint(len(_dataset_shsy5y_val['images']))\nprint(len(_dataset_shsy5y_test['images']))","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:33:06.533746Z","iopub.execute_input":"2021-11-25T04:33:06.534049Z","iopub.status.idle":"2021-11-25T04:33:06.53982Z","shell.execute_reply.started":"2021-11-25T04:33:06.534016Z","shell.execute_reply":"2021-11-25T04:33:06.538757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(_dataset['images'][0])","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.572724Z","iopub.status.idle":"2021-11-25T04:23:30.573208Z","shell.execute_reply.started":"2021-11-25T04:23:30.572943Z","shell.execute_reply":"2021-11-25T04:23:30.572978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_dataset['annotations']['2']['segmentation']","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.575021Z","iopub.status.idle":"2021-11-25T04:23:30.57554Z","shell.execute_reply.started":"2021-11-25T04:23:30.575261Z","shell.execute_reply":"2021-11-25T04:23:30.575286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_dataset['annotations']['2']['bbox']","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.577152Z","iopub.status.idle":"2021-11-25T04:23:30.577648Z","shell.execute_reply.started":"2021-11-25T04:23:30.57739Z","shell.execute_reply":"2021-11-25T04:23:30.577415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_check = _dataset['images'][2]\n_check","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.579197Z","iopub.status.idle":"2021-11-25T04:23:30.579731Z","shell.execute_reply.started":"2021-11-25T04:23:30.579456Z","shell.execute_reply":"2021-11-25T04:23:30.579483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tiffiles[_check['file_name']]","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.581666Z","iopub.status.idle":"2021-11-25T04:23:30.58219Z","shell.execute_reply.started":"2021-11-25T04:23:30.581906Z","shell.execute_reply":"2021-11-25T04:23:30.581932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_dataset_anno = _dataset['annotations']\n_image_id = _check['id']\n\n_image = tifffile.imread(tiffiles[_check['file_name']])","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.584121Z","iopub.status.idle":"2021-11-25T04:23:30.584747Z","shell.execute_reply.started":"2021-11-25T04:23:30.584431Z","shell.execute_reply":"2021-11-25T04:23:30.58446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"allMask = np.zeros((520, 704))\nfor key in _dataset_anno.keys():\n    if _dataset_anno[key]['image_id'] == _image_id :\n        rle = MeaskUtils.frPoly(_dataset_anno[key]['segmentation'],520,704 )\n        mask = MeaskUtils.decode(rle)\n        allMask += mask[:,:,0]\n\nplt.figure(figsize=(20,10))\nplt.subplot(1,3,1)\nplt.imshow(_image) \n\nplt.subplot(1,3,2)\nplt.imshow(allMask == 1, cmap='jet', alpha=0.5) \n\nplt.subplot(1,3,3)\nmerged = cv.addWeighted(allMask, 0.75, allMask, 0.25, 0.0,)\nplt.imshow(merged)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.586113Z","iopub.status.idle":"2021-11-25T04:23:30.586611Z","shell.execute_reply.started":"2021-11-25T04:23:30.58635Z","shell.execute_reply":"2021-11-25T04:23:30.586375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Semisupervised Images","metadata":{}},{"cell_type":"code","source":"SEMI_DIR = '../input/sartorius-cell-instance-segmentation/train_semi_supervised'","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.588027Z","iopub.status.idle":"2021-11-25T04:23:30.588519Z","shell.execute_reply.started":"2021-11-25T04:23:30.588242Z","shell.execute_reply":"2021-11-25T04:23:30.588268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unsupervised_files = []\nunsupervised_files_path = []\nfor filename in os.listdir(SEMI_DIR):\n    unsupervised_files_path.append(os.path.join(SEMI_DIR, filename))\n    unsupervised_files.append(filename)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.590914Z","iopub.status.idle":"2021-11-25T04:23:30.591499Z","shell.execute_reply.started":"2021-11-25T04:23:30.591153Z","shell.execute_reply":"2021-11-25T04:23:30.591181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"semi_df = pd.DataFrame()\n\nsemi_df[\"cell_type\"] = [x.split(\"[\", 1)[0] for x in unsupervised_files]\nsemi_df[\"compound\"] = [x.split(\"]\", 1)[0].split(\"[\", 1)[-1] for x in unsupervised_files]\nsemi_df[\"img_path\"] = unsupervised_files_path","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.592702Z","iopub.status.idle":"2021-11-25T04:23:30.593195Z","shell.execute_reply.started":"2021-11-25T04:23:30.592926Z","shell.execute_reply":"2021-11-25T04:23:30.592953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(semi_df, \"cell_type\", color=\"compound\")\nfig.show()\n\nfig = px.histogram(semi_df, \"compound\", color=\"cell_type\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.59487Z","iopub.status.idle":"2021-11-25T04:23:30.595392Z","shell.execute_reply.started":"2021-11-25T04:23:30.595096Z","shell.execute_reply":"2021-11-25T04:23:30.595123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"semi_df","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.59853Z","iopub.status.idle":"2021-11-25T04:23:30.598904Z","shell.execute_reply.started":"2021-11-25T04:23:30.59872Z","shell.execute_reply":"2021-11-25T04:23:30.598738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,26))\nfor i, img_path in zip(range(15), semi_df.img_path.to_list()):\n    plt.subplot(5,3,i+1)\n    plt.imshow((255-np.asarray(ImageEnhance.Contrast(Image.fromarray(cv.imread(img_path))).enhance(16))), cmap=\"inferno\")\n    plt.axis(False)\n    plt.title(img_path.rsplit(\"/\", 1)[-1].rsplit(\".\", 1)[0], fontweight=\"bold\")\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-25T04:23:30.600536Z","iopub.status.idle":"2021-11-25T04:23:30.600913Z","shell.execute_reply.started":"2021-11-25T04:23:30.600736Z","shell.execute_reply":"2021-11-25T04:23:30.600755Z"},"trusted":true},"execution_count":null,"outputs":[]}]}