{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Data Exploration\n\n* The Dataset belongs to the following Kaggle Competition :\n* https://www.kaggle.com/c/sartorius-cell-instance-segmentation/data\n\n","metadata":{}},{"cell_type":"code","source":"!pip install pycocotools","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-03T22:54:37.986053Z","iopub.execute_input":"2022-01-03T22:54:37.986415Z","iopub.status.idle":"2022-01-03T22:54:47.052669Z","shell.execute_reply.started":"2022-01-03T22:54:37.986381Z","shell.execute_reply":"2022-01-03T22:54:47.051473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom IPython import display\nfrom PIL import Image, ImageEnhance\nimport matplotlib.image as mpimg\nimport json,itertools\nimport skimage.io as io\nfrom pathlib import Path\nfrom pycocotools.coco import COCO\n","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:36:58.284341Z","iopub.execute_input":"2022-01-03T22:36:58.284800Z","iopub.status.idle":"2022-01-03T22:36:59.116082Z","shell.execute_reply.started":"2022-01-03T22:36:58.284764Z","shell.execute_reply":"2022-01-03T22:36:59.114998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Dataframe\n* Previewing the training data\n* Creating a function to display an image from the files provided ","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf.head()\n#print(df.loc[0, 'annotation'])\n","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:36:59.117729Z","iopub.execute_input":"2022-01-03T22:36:59.118192Z","iopub.status.idle":"2022-01-03T22:36:59.807809Z","shell.execute_reply.started":"2022-01-03T22:36:59.118142Z","shell.execute_reply":"2022-01-03T22:36:59.806602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def displayImage(id,folder):\n    #'../input/sartorius-cell-instance-segmentation/train\n    \n    #Image.open(dataDir/img['file_name'])\n    imageName = id + '.png'\n    folder = folder\n    directory = \"../input/sartorius-cell-instance-segmentation/\"+folder+\"/\"\n    imgPath = directory + imageName\n    img = mpimg.imread(imgPath)\n    imgplot = plt.imshow(img,cmap=\"gray\")\n\n\ndisplayImage('0a6ecc5fe78a','train')","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:36:59.810125Z","iopub.execute_input":"2022-01-03T22:36:59.810387Z","iopub.status.idle":"2022-01-03T22:37:00.210125Z","shell.execute_reply.started":"2022-01-03T22:36:59.810357Z","shell.execute_reply":"2022-01-03T22:37:00.208743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# 2. Data Preprocessing \n\nCOCO is an object segmentation and object detection and captioning dataset which is perfect to be used to look at cells and is the format in which Detectron2 accepts the data to perform image segmentation. The focus of this part of the notebook will be to ensure that the data is correctly organized in the format accepted by COCO.\n\n* https://www.kaggle.com/coldfir3/efficient-coco-dataset-generator\n* https://towardsdatascience.com/how-to-work-with-object-detection-datasets-in-coco-format-9bf4fb5848a4\n\nMore links about COCO Dataset\n\n* https://cocodataset.org/#home\n* https://github.com/cocodataset/cocoapi \n* https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#Expected output structure\n'''\n{\n    \"categories\": [\n        {\n            \"name\": \"shsy5y\",\n            \"id\": 1\n        },\n        {\n            \"name\": \"astro\",\n            \"id\": 2\n        },\n        {\n            \"name\": \"cort\",\n            \"id\": 3\n        }\n    ],\n    \"images\": [\n        {\n            \"id\": \"0030fd0e6378\",\n            \"width\": 704,\n            \"height\": 520,\n            \"file_name\": \"train/0030fd0e6378.png\"\n        },\n    ],\n    \"annotations\": [\n        {\n            \"segmentation\": {\n                \"counts\": [\n                    299687,\n                    7,\n                    513,\n                    19,\n                    501,\n                    25,\n                    495,\n                    .\n                    .\n                    .\n                ], \n                \"size\": [\n                    520,\n                    704\n                ]\n            }, \n            \"box\": [\n                679,\n                149,\n                25,\n                37\n            ], \n            \"area\": 420, \n            \"image_id\": \"ffc2ead3e8cc\", \n            \"category_id\": 0,\n            \"iscrowd\": 0, \n            \"id\": 73506\n        }\n    ]\n}\n'''","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:00.211943Z","iopub.execute_input":"2022-01-03T22:37:00.213099Z","iopub.status.idle":"2022-01-03T22:37:00.221537Z","shell.execute_reply.started":"2022-01-03T22:37:00.213050Z","shell.execute_reply":"2022-01-03T22:37:00.220774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create a mask on the area where a cell exists\n\n* The purpose of this function is to return a mask which is identified as 1 or 0 which is identified as background\n* The rle is decoded by taking the start and end of a pixel and filling the start to end with 1s to denote the cell present   \n    \nSample Annotation input\n\n118145 6 118849 7 119553 8 120257 8 120961 9 121665 10 122369 12 123074 13 123778 14 124482 15 125186 16 125890 17 126594 18 127298 19 128002 20 128706 21 129410 22 130114 23 130818 24 131523 24 132227 25 132931 25 133635 24 134339 24 135043 23 135748 21 136452 19 137157 16 137864 11 138573 4\n","metadata":{}},{"cell_type":"code","source":"def covertRLE(annotation,height,width):\n    #convert String to array\n    rle = annotation.split()\n    #print(rle)\n    starts_of_pixel, lengths_of_pixel= [np.asarray(x, dtype=int) for x in (rle[0:][::2],rle[1:][::2])]\n    #print(lengths_of_pixel)\n    #print(starts_of_pixel)\n    #move start over 1 because numpy indexing starts at 0\n    starts_of_pixel -= 1\n    ends_of_pixel =  starts_of_pixel + lengths_of_pixel\n    #create empty matrix filled with 0s to store final values\n    img = np.zeros(width*height, dtype=np.uint8)\n    #identify these regions as 1 \n    for start, end in zip(starts_of_pixel, ends_of_pixel):\n        img[start:end] = 1\n    #create a 2D Array with height as number of rows and width as columns\n    return img.reshape(height,width)\n    \n\nsingleCell = df.loc[2534, 'annotation']\n#print(singleCell)\n#print(df.loc[2534, 'id'])\nheight = 520\nwidth = 704\nconvertedImg = covertRLE(singleCell,height,width)\n\nplt.imshow(convertedImg,cmap='gray');\n#print(convertedImg)\n#ansArr = createImgMask(singleAnn,width,height)\n#print(ansArr.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:00.222923Z","iopub.execute_input":"2022-01-03T22:37:00.223174Z","iopub.status.idle":"2022-01-03T22:37:00.530208Z","shell.execute_reply.started":"2022-01-03T22:37:00.223142Z","shell.execute_reply":"2022-01-03T22:37:00.529007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Count the number of pixels that have cells\n* This function will return the data in the form of number of 0s followed by number of 1s\n\n https://newbedev.com/encode-numpy-array-using-uncompressed-rle-for-coco-dataset","metadata":{}},{"cell_type":"code","source":"def getCountsOfMask(binary_mask):\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    counts_obj = rle.get('counts')\n    #print( counts_obj)\n    \n    #Flatten the file \n    for i, (value, elements) in enumerate(itertools.groupby(binary_mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n             counts_obj.append(0)\n        counts_obj.append(len(list(elements)))\n\n    return rle\n\n\ntest_binary_mask =  np.array([[  0,   0,   0,   0,   0,  1,  0,   0,   0,   1]], dtype=np.uint8)\n\n#{'counts': [5, 1, 3, 1], 'size': [1, 10]}\n\n'''\ntest_binary_mask =  np.array([[  0,   0,   0,   0,   0,   0,   0,   0,   0,   0],\n                                     [  0,   0,   0,   0,   0,   0,   0,   0,   0,   0],\n                                     [  0,   0,   0,   0,   0,   1,   1,   1,   0,   0],\n                                     [  0,   0,   0,   0,   0,   1,   1,   1,   0,   0],\n                                     [  0,   0,   0,   0,   0,   1,   1,   1,   0,   0],\n                                     [  0,   0,   0,   0,   0,   1,   1,   1,   0,   0],\n                                     [  1,   0,   0,   0,   0,   0,   0,   0,   0,   0],\n                                     [  0,   0,   0,   0,   0,   0,   0,   0,   0,   0],\n                                     [  0,   0,   0,   0,   0,   0,   0,   0,   0,   0]], dtype=np.uint8)\n\n'''\n\nprint(getCountsOfMask(test_binary_mask))","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:00.531741Z","iopub.execute_input":"2022-01-03T22:37:00.531977Z","iopub.status.idle":"2022-01-03T22:37:00.542906Z","shell.execute_reply.started":"2022-01-03T22:37:00.531948Z","shell.execute_reply":"2022-01-03T22:37:00.542127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert the Dataframe into a readable .json file\n\n* This function will create a category map of all the types of cells that are \npresent in the training file.\n","metadata":{}},{"cell_type":"code","source":"def createCategoryMapByCellType(train_df):\n    categoryDict = []\n    cellCategories = train_df['cell_type'].values\n    uniqueCellCategories = np.unique(cellCategories)\n    #print(uniqueCellCategories)\n    for i in range(0,len(uniqueCellCategories)):\n        cellCategory = uniqueCellCategories[i]\n        if cellCategory not in categoryDict:\n            newCat = {\n                'name':'',\n                'id': ''\n            }\n            newCat['name'] = cellCategory\n            newCat['id'] = i+1\n            categoryDict.append(newCat)  \n        #print(categoryDict)\n    return categoryDict\n\ndef getCategoryID(cat_name,all_cats):\n    for idx, cat in enumerate(all_cats):\n        if cat['name'] == cat_name:\n            return cat['id']\n    \nsample_rows = df.iloc[73506:73512,:]\ncat_map = createCategoryMapByCellType(sample_rows)\nget_id = getCategoryID('astro',cat_map)\n#print(sample_rows)\n#print(cat_map)\n#print(get_id)","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:00.544145Z","iopub.execute_input":"2022-01-03T22:37:00.545093Z","iopub.status.idle":"2022-01-03T22:37:00.559178Z","shell.execute_reply.started":"2022-01-03T22:37:00.545056Z","shell.execute_reply":"2022-01-03T22:37:00.558180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function that builds the COCO .json file","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\n\ndef createCocoStructure(train_df):\n    all_cell_arr = {\n        'categories' : [],\n        'images' : [],\n        'annotations':[]\n    }\n    cat_map = createCategoryMapByCellType(train_df)\n    all_cell_arr['categories'] = cat_map\n   \n    for idx, row in tqdm(train_df.iterrows()):        \n        cell_category_name = row.cell_type\n        cell_category_id =  getCategoryID(cell_category_name,cat_map)\n        mask = covertRLE(row.annotation, row.height, row.width)\n        enc_binary_ann_counts = getCountsOfMask(mask)\n        row_id = row.id\n        img_path = 'train/'+row_id+'.png'\n        #create the image obj \n        img_obj = {\n            'id':row_id,\n            'width' : row.width,\n            'height' : row.height,\n            'file_name': img_path\n        }\n        all_cell_arr['images'].append(img_obj)\n        #create box around pixel\n        ys, xs = np.where(mask)\n        x1, x2 = min(xs), max(xs)\n        y1, y2 = min(ys), max(ys)        \n        ann_obj = {\n            'segmentation' :  enc_binary_ann_counts,\n            'bbox' : [int(x1),int(y1),int(x2-x1+1),int(y2-y1+1)],\n            'area':int(np.sum(mask)),\n            'image_id':row.id,\n            'category_id': cell_category_id,\n            'iscrowd' : 0,\n            'id' :idx\n        }\n        all_cell_arr['annotations'].append(ann_obj)\n        \n    return all_cell_arr\n    \n#sample_rows = df.iloc[73506:73512,:]\n#print(sample_rows)\n#root = createCocoStructure(sample_rows) \n\n#Save the test file\n#with open('annotations_sample_test.json', 'w', encoding='utf-8') as f:\n#    json.dump(root, f, ensure_ascii=True, indent=4)\n#print(root)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:00.560714Z","iopub.execute_input":"2022-01-03T22:37:00.561741Z","iopub.status.idle":"2022-01-03T22:37:00.645605Z","shell.execute_reply.started":"2022-01-03T22:37:00.561702Z","shell.execute_reply":"2022-01-03T22:37:00.644800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the Data to Train and Test Sets \nSince there is 73586 rows the data can be split as \n1. 80% Training - 58868\n2. 20% Testing - 14717\n","metadata":{}},{"cell_type":"code","source":" \nfrom sklearn.model_selection import train_test_split\ndf_train, df_test = train_test_split(df, test_size=0.20, stratify=df['cell_type'])\ndf_train.head()\n#print(df_train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:00.648839Z","iopub.execute_input":"2022-01-03T22:37:00.649400Z","iopub.status.idle":"2022-01-03T22:37:01.895194Z","shell.execute_reply.started":"2022-01-03T22:37:00.649359Z","shell.execute_reply":"2022-01-03T22:37:01.894025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()\n#print(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:01.896919Z","iopub.execute_input":"2022-01-03T22:37:01.897222Z","iopub.status.idle":"2022-01-03T22:37:01.914512Z","shell.execute_reply.started":"2022-01-03T22:37:01.897190Z","shell.execute_reply":"2022-01-03T22:37:01.913129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createAndSaveFile(df,fileName):\n    fileNameJson = fileName+ '.json' \n    root = createCocoStructure(df) \n    with open(fileNameJson, 'w', encoding='utf-8') as f:\n        json.dump(root, f, ensure_ascii=True, indent=4)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:01.916330Z","iopub.execute_input":"2022-01-03T22:37:01.916812Z","iopub.status.idle":"2022-01-03T22:37:01.926290Z","shell.execute_reply.started":"2022-01-03T22:37:01.916764Z","shell.execute_reply":"2022-01-03T22:37:01.925245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#createAndSaveFile(df_train,'annotations_train_final')\n#createAndSaveFile(df_test,'annotations_test_final')","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:01.928541Z","iopub.execute_input":"2022-01-03T22:37:01.929015Z","iopub.status.idle":"2022-01-03T22:37:01.945125Z","shell.execute_reply.started":"2022-01-03T22:37:01.928967Z","shell.execute_reply":"2022-01-03T22:37:01.944095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#View Train file - increase number of lines to see more \n!head -n 10 ../input/annotations-test-final/annotations_test_final.json","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:01.946247Z","iopub.execute_input":"2022-01-03T22:37:01.946506Z","iopub.status.idle":"2022-01-03T22:37:02.755999Z","shell.execute_reply.started":"2022-01-03T22:37:01.946476Z","shell.execute_reply":"2022-01-03T22:37:02.754943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensure coco dataset is able to categorize images\n* The data in the dataset only shows annotations of a sincle cell, which is why the iscrowd attribute was set to 0\n* Now we can see if Coco is able to identify the other cell pixels based on the data of that one cell segmentation\n* Loading 10 images to check the box outlines\n\n# The following API functions are defined:\n  COCO       - COCO api class that loads COCO annotation file and prepare data structures.\n* decodeMask - Decode binary mask M encoded via run-length encoding.\n* encodeMask - Encode binary mask M using run-length encoding.\n* getAnnIds  - Get ann ids that satisfy given filter conditions.\n* getCatIds  - Get cat ids that satisfy given filter conditions.\n* getImgIds  - Get img ids that satisfy given filter conditions.\n* loadAnns   - Load anns with the specified ids.\n* loadCats   - Load cats with the specified ids.\n* loadImgs   - Load imgs with the specified ids.\n* annToMask  - Convert segmentation in an annotation to binary mask.\n* showAnns   - Display the specified annotations.\n* loadRes    - Load algorithm results and create API for accessing them.\n* download   - Download COCO images from mscoco.org server.\n\nhttps://github.com/cocodataset/cocoapi/blob/master/PythonAPI/pycocotools/coco.py","metadata":{}},{"cell_type":"markdown","source":"","metadata":{"execution":{"iopub.status.busy":"2021-12-29T00:12:57.101043Z","iopub.execute_input":"2021-12-29T00:12:57.101498Z","iopub.status.idle":"2021-12-29T00:13:01.509126Z","shell.execute_reply.started":"2021-12-29T00:12:57.10146Z","shell.execute_reply":"2021-12-29T00:13:01.507775Z"}}},{"cell_type":"code","source":" \ndataDir=Path('../input/sartorius-cell-instance-segmentation')\ntrFile = Path('../input/annotations-train-final/annotations_train_final.json')\ncoco = COCO(trFile)\nimgIds = coco.getImgIds()\nimgs = coco.loadImgs(imgIds)\n    \nimgs = coco.loadImgs(imgIds[-10:])\n_,axs = plt.subplots(len(imgs),2,figsize=(40,15 * len(imgs)))\nfor img, ax in zip(imgs, axs):\n    I = io.imread(dataDir/img['file_name'])\n    annIds = coco.getAnnIds(imgIds=[img['id']])\n    anns = coco.loadAnns(annIds)\n    ax[0].imshow(I)\n    ax[1].imshow(I)\n    plt.sca(ax[1])\n    coco.showAnns(anns, draw_bbox=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-03T22:37:02.758285Z","iopub.execute_input":"2022-01-03T22:37:02.758724Z","iopub.status.idle":"2022-01-03T22:37:55.801868Z","shell.execute_reply.started":"2022-01-03T22:37:02.758655Z","shell.execute_reply":"2022-01-03T22:37:55.799516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}