{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install git+git://github.com/waspinator/pycococreator.git@0.2.0\n!pip install git+git://github.com/waspinator/coco.git@2.1.0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nimport json\nimport os\nimport re\nfrom glob import glob\nimport fnmatch\nfrom PIL import Image\nimport numpy as np\nfrom pycococreatortools import pycococreatortools\nimport pandas as pd\n\nfrom skimage.io import imread\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\n\ndataset_train = '../input/sartorius-cell-instance-segmentation/train'\ncsv_train = '../input/sartorius-cell-instance-segmentation/train.csv'\nIMAGE_DIR = dataset_train\n\ndf = pd.read_csv(csv_train )  # read csv file\n","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:00:20.36427Z","iopub.execute_input":"2021-10-19T13:00:20.364658Z","iopub.status.idle":"2021-10-19T13:00:22.188123Z","shell.execute_reply.started":"2021-10-19T13:00:20.364608Z","shell.execute_reply":"2021-10-19T13:00:22.187479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INFO = {\n    \"description\": \"Kaggle Dataset\",\n    \"url\": \"https://github.com/pmj110119\",\n    \"version\": \"0.1.0\",\n    \"year\": 2021,\n    \"contributor\": \"pmj110119\",\n    \"date_created\": datetime.datetime.utcnow().isoformat(' ')\n}\n\nLICENSES = [\n    {\n        \"id\": 1,\n        \"name\": \"Attribution-NonCommercial-ShareAlike License\",\n        \"url\": \"http://creativecommons.org/licenses/by-nc-sa/2.0/\"\n    }\n]\n\nCATEGORIES = [\n    {\n        'id': 1,\n        'name': 'cell',\n        'supercategory': 'cell',\n    },\n]","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:00:22.189522Z","iopub.execute_input":"2021-10-19T13:00:22.189964Z","iopub.status.idle":"2021-10-19T13:00:22.195369Z","shell.execute_reply.started":"2021-10-19T13:00:22.189924Z","shell.execute_reply":"2021-10-19T13:00:22.194514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode(mask_rle, shape=(520, 704)):\n    s = mask_rle.split()\n    starts =  np.asarray(s[0::2], dtype=int)\n    lengths = np.asarray(s[1::2], dtype=int)\n\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction\n\ndef rle_decode(mask_rle, shape=(520, 704), color=1):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    # Split the string by space, then convert it into a integer array\n    s = np.array(mask_rle.split(), dtype=int)\n\n    # Every even value is the start, every odd value is the \"run\" length\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n\n    # The image image is actually flattened since RLE is a 1D \"run\"\n    if len(shape)==3:\n        h, w, d = shape\n        img = np.zeros((h * w, d), dtype=np.float32)\n    else:\n        h, w = shape\n        img = np.zeros((h * w,), dtype=np.float32)\n\n    # The color here is actually just any integer you want!\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = color\n        \n    # Don't forget to change the image back to the original shape\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:00:22.197572Z","iopub.execute_input":"2021-10-19T13:00:22.19797Z","iopub.status.idle":"2021-10-19T13:00:22.212126Z","shell.execute_reply.started":"2021-10-19T13:00:22.19793Z","shell.execute_reply":"2021-10-19T13:00:22.211267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 最终放进json文件里的字典\ncoco_output = {\n    \"info\": INFO,\n    \"licenses\": LICENSES,\n    \"categories\": CATEGORIES,\n    \"images\": [],   # 放一个空列表占位置，后面再append\n    \"annotations\": []\n}\n\nimage_id = 1\nsegmentation_id = 1\n\nimage_paths = glob(os.path.join(IMAGE_DIR,'*.png'))\n# 遍历每一张图片\nfor image_path in tqdm(image_paths,total=len(image_paths)):\n    if image_id > 5:    # delete this when used\n        break\n    # 提取图片信息\n    image = Image.open(image_path)\n    image_name = os.path.basename(image_path)   # 不需要具体的路径，只要图片文件名\n    image_info = pycococreatortools.create_image_info(\n        image_id, image_name, image.size)\n    coco_output[\"images\"].append(image_info)\n\n    # 内层循环是mask，把每一张图片的mask搜索出来\n    rle_masks = df.loc[df['id'] == image_name[:-4], 'annotation'].tolist()\n    for index in range(len(rle_masks)):\n        binary_mask = rle_decode(rle_masks[index])\n        class_id = 1    # 所有图片的类别都是1，ship\n        category_info = {'id': class_id, 'is_crowd': 0}\n        annotation_info = pycococreatortools.create_annotation_info(\n            segmentation_id, image_id, category_info, binary_mask,\n            image.size, tolerance=1)\n    \n        # save result\n        coco_output[\"annotations\"].append(annotation_info)\n        \n        # 无论标注是否被写入数据集，均分配一个编号\n        segmentation_id = segmentation_id + 1   \n        \n    image_id = image_id + 1\n    \nwith open('instances_cell_train2021.json', 'w') as output_json_file:\n    json.dump(coco_output, output_json_file,indent=4)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:00:22.213781Z","iopub.execute_input":"2021-10-19T13:00:22.21401Z","iopub.status.idle":"2021-10-19T13:00:37.673399Z","shell.execute_reply.started":"2021-10-19T13:00:22.213986Z","shell.execute_reply":"2021-10-19T13:00:37.672468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check COCO","metadata":{"execution":{"iopub.status.busy":"2021-10-19T05:03:02.61984Z","iopub.execute_input":"2021-10-19T05:03:02.620349Z","iopub.status.idle":"2021-10-19T05:03:02.623486Z","shell.execute_reply.started":"2021-10-19T05:03:02.620286Z","shell.execute_reply":"2021-10-19T05:03:02.622673Z"}}},{"cell_type":"code","source":"from pycocotools.coco import COCO\n\nannFile='instances_cell_train2021.json'\ncoco = COCO(annFile)","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:00:37.674819Z","iopub.execute_input":"2021-10-19T13:00:37.675144Z","iopub.status.idle":"2021-10-19T13:00:37.820846Z","shell.execute_reply.started":"2021-10-19T13:00:37.675102Z","shell.execute_reply":"2021-10-19T13:00:37.819962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('categories:',coco.dataset['categories'])\nprint('image nums:',len(coco.dataset['images']))\nprint('annotation nums:',len(coco.dataset['annotations']))","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:00:38.2029Z","iopub.execute_input":"2021-10-19T13:00:38.20319Z","iopub.status.idle":"2021-10-19T13:00:38.209275Z","shell.execute_reply.started":"2021-10-19T13:00:38.203159Z","shell.execute_reply":"2021-10-19T13:00:38.208555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skimage import io\n\n# select one at random\nimgIds = coco.getImgIds(catIds=[1])\nimg = coco.loadImgs(imgIds[np.random.randint(0,len(imgIds))])[0]\nprint('file_name:',img['file_name'])\n\n# load and display origin image\nI = io.imread('%s/%s'%(IMAGE_DIR,img['file_name']))\nplt.axis('off')\nplt.imshow(I)\nplt.show()\n\n# load and display instance annotations\nplt.imshow(I); plt.axis('off')\nannIds = coco.getAnnIds(imgIds=img['id'], catIds=[1], iscrowd=None)\nanns = coco.loadAnns(annIds)\ncoco.showAnns(anns)","metadata":{"execution":{"iopub.status.busy":"2021-10-19T13:01:44.164632Z","iopub.execute_input":"2021-10-19T13:01:44.165028Z","iopub.status.idle":"2021-10-19T13:01:44.596043Z","shell.execute_reply.started":"2021-10-19T13:01:44.164991Z","shell.execute_reply":"2021-10-19T13:01:44.595222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}