{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport zipfile\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-23T18:54:02.059896Z","iopub.execute_input":"2021-11-23T18:54:02.060305Z","iopub.status.idle":"2021-11-23T18:54:02.732769Z","shell.execute_reply.started":"2021-11-23T18:54:02.060245Z","shell.execute_reply":"2021-11-23T18:54:02.731377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*References:*\n\n* www.kaggle.com/ihelon/cell-segmentation-run-length-decoding\n* www.kaggle.com/inversion/run-length-decoding-quick-start\n* www.kaggle.com/ihelon/cell-segmentation-run-length-decoding\n* www.kaggle.com/robertlangdonvinci/sartorius-cell-segmentation-data-gen/notebook\n","metadata":{}},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/sartorius-cell-instance-segmentation/'\nTRAIN_CSV = os.path.join(ROOT_DIR, 'train.csv')\nwith open(TRAIN_CSV, 'r') as f:\n     data_df = pd.read_csv(f, delimiter=',')\ndata_df[:10]\n","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:02.738896Z","iopub.execute_input":"2021-11-23T18:54:02.7392Z","iopub.status.idle":"2021-11-23T18:54:03.204879Z","shell.execute_reply.started":"2021-11-23T18:54:02.739167Z","shell.execute_reply":"2021-11-23T18:54:03.203883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mpimg\nimport matplotlib.pyplot as plt\nIMAGE_DIR = os.path.join(ROOT_DIR, 'train')\ncount = 0\nfor dirname, _, filenames in os.walk(IMAGE_DIR):\n    for filename in filenames:\n        img = mpimg.imread(os.path.join(dirname, filename))\n        imgplot = plt.imshow(img)\n        print(img.shape)\n        plt.show()\n        count += 1\n        if count == 3: break","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:03.206524Z","iopub.execute_input":"2021-11-23T18:54:03.206765Z","iopub.status.idle":"2021-11-23T18:54:04.116235Z","shell.execute_reply.started":"2021-11-23T18:54:03.20674Z","shell.execute_reply":"2021-11-23T18:54:04.11493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref: https://www.kaggle.com/inversion/run-length-decoding-quick-start\ndef rle_decode(mask_rle, shape, color=1):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height, width, channels) of array to return \n    color: color for the mask\n    Returns numpy array (mask)\n\n    '''\n    s = mask_rle.split()\n    \n    starts = list(map(lambda x: int(x) - 1, s[0::2]))\n    lengths = list(map(int, s[1::2]))\n    ends = [x + y for x, y in zip(starts, lengths)]\n    \n    img = np.zeros((shape[0] * shape[1], shape[2]), dtype=np.float32)\n            \n    for start, end in zip(starts, ends):\n        img[start : end] = color\n    \n    return img.reshape(shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:04.11962Z","iopub.execute_input":"2021-11-23T18:54:04.120018Z","iopub.status.idle":"2021-11-23T18:54:04.12836Z","shell.execute_reply.started":"2021-11-23T18:54:04.119969Z","shell.execute_reply":"2021-11-23T18:54:04.127343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref: www.kaggle.com/ihelon/cell-segmentation-run-length-decoding\n# ref: www.kaggle.com/robertlangdonvinci/sartorius-cell-segmentation-data-gen/notebook\ndef plot_masks(dataframe, image_id, colors=True):\n    labels = dataframe[dataframe[\"id\"] == image_id][\"annotation\"].tolist()\n\n    if colors:\n        mask = np.zeros((520, 704, 3))\n        for label in labels:\n            mask += rle_decode(label, shape=(520, 704, 3), color=np.random.rand(3))\n    else:\n        mask = np.zeros((520, 704, 1))\n        for label in labels:\n            mask += rle_decode(label, shape=(520, 704, 1))\n    mask = mask.clip(0, 1)\n\n    image = cv2.imread(os.path.join(ROOT_DIR, f\"train/{image_id}.png\"))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    plt.figure(figsize=(15, 8))\n    plt.subplot(131)\n    plt.imshow(image)\n    plt.title('raw image')\n    plt.axis(\"off\")\n    plt.subplot(132)\n    plt.imshow(image)\n    plt.imshow(mask, alpha=0.6)\n    plt.title('image + mask')\n    plt.axis(\"off\")\n    plt.subplot(133)\n    plt.imshow(mask)\n    plt.title('mask only')\n    plt.axis(\"off\")\n    plt.tight_layout()\n    plt.show();","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:04.130279Z","iopub.execute_input":"2021-11-23T18:54:04.130552Z","iopub.status.idle":"2021-11-23T18:54:04.148456Z","shell.execute_reply.started":"2021-11-23T18:54:04.130514Z","shell.execute_reply":"2021-11-23T18:54:04.147431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_masks(data_df, \"0030fd0e6378\", colors=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:04.149881Z","iopub.execute_input":"2021-11-23T18:54:04.150151Z","iopub.status.idle":"2021-11-23T18:54:05.705711Z","shell.execute_reply.started":"2021-11-23T18:54:04.150113Z","shell.execute_reply":"2021-11-23T18:54:05.704778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=data_df.cell_type)","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:05.706977Z","iopub.execute_input":"2021-11-23T18:54:05.707256Z","iopub.status.idle":"2021-11-23T18:54:05.878715Z","shell.execute_reply.started":"2021-11-23T18:54:05.707223Z","shell.execute_reply":"2021-11-23T18:54:05.87807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_type = data_df['cell_type'].unique();cell_type","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:05.880791Z","iopub.execute_input":"2021-11-23T18:54:05.881481Z","iopub.status.idle":"2021-11-23T18:54:05.899794Z","shell.execute_reply.started":"2021-11-23T18:54:05.881425Z","shell.execute_reply":"2021-11-23T18:54:05.898622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:05.90728Z","iopub.execute_input":"2021-11-23T18:54:05.907836Z","iopub.status.idle":"2021-11-23T18:54:05.926646Z","shell.execute_reply.started":"2021-11-23T18:54:05.907777Z","shell.execute_reply":"2021-11-23T18:54:05.925765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df['cell_type'].replace({'shsy5y':1,'astro':2,'cort':3},inplace=True)\ndata_df['cell_type'] = pd.to_numeric(data_df['cell_type'])\ndata_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:05.928131Z","iopub.execute_input":"2021-11-23T18:54:05.928475Z","iopub.status.idle":"2021-11-23T18:54:05.976872Z","shell.execute_reply.started":"2021-11-23T18:54:05.928431Z","shell.execute_reply":"2021-11-23T18:54:05.975948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_group = data_df.groupby('id')\ndata_group.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:05.978212Z","iopub.execute_input":"2021-11-23T18:54:05.978709Z","iopub.status.idle":"2021-11-23T18:54:06.009196Z","shell.execute_reply.started":"2021-11-23T18:54:05.978672Z","shell.execute_reply":"2021-11-23T18:54:06.008459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_mask(img_id, dataframe, color=1):\n    temp = dataframe.get_group(img_id)\n    temp_annot = temp.loc[:,'annotation'].tolist()\n    mask = np.zeros((520, 704, 1))\n    for label in temp_annot:\n        mask += rle_decode(label, shape=(520, 704, 1))\n    mask = mask.clip(0, 1)\n    mask[mask==1] = color\n    return mask\n\n","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:06.010321Z","iopub.execute_input":"2021-11-23T18:54:06.011128Z","iopub.status.idle":"2021-11-23T18:54:06.019591Z","shell.execute_reply.started":"2021-11-23T18:54:06.011057Z","shell.execute_reply":"2021-11-23T18:54:06.018554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import stats\nctype_df = data_df[['id','cell_type']].groupby('id').agg(lambda x:stats.mode(np.array(x))[0]).reset_index()\nctype_df[:10]","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:06.021669Z","iopub.execute_input":"2021-11-23T18:54:06.022378Z","iopub.status.idle":"2021-11-23T18:54:06.131838Z","shell.execute_reply.started":"2021-11-23T18:54:06.02233Z","shell.execute_reply":"2021-11-23T18:54:06.130997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"OUT_TRAIN = 'TrainMask.zip'\nfiles = np.array(list(zip(ctype_df['id'],ctype_df['cell_type'])))","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:06.133236Z","iopub.execute_input":"2021-11-23T18:54:06.134008Z","iopub.status.idle":"2021-11-23T18:54:06.141259Z","shell.execute_reply.started":"2021-11-23T18:54:06.133958Z","shell.execute_reply":"2021-11-23T18:54:06.140321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile(OUT_TRAIN, 'w') as img_out:\n    for idx in tqdm(range(0,len(files))):\n        temp_mask = build_mask(files[idx][0],data_group, color=int(files[idx][1]))\n        M = temp_mask.shape[0]//2\n        N = temp_mask.shape[1]//2\n        tiles = [temp_mask[x:x+M,y:y+N] for x in range(0,temp_mask.shape[0],M) for y in range(0,temp_mask.shape[1],N)]\n        for j in range(4):\n            mask1 = tiles[j]\n            mask1 = cv2.imencode('.png',mask1)[1]\n            img_out.writestr(files[idx][0] + f'_{j}_mask.png', mask1)","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:06.142742Z","iopub.execute_input":"2021-11-23T18:54:06.14322Z","iopub.status.idle":"2021-11-23T18:54:44.381683Z","shell.execute_reply.started":"2021-11-23T18:54:06.143169Z","shell.execute_reply":"2021-11-23T18:54:44.380067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"OUT_TRAIN = 'TrainImage.zip'","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:44.383616Z","iopub.execute_input":"2021-11-23T18:54:44.384039Z","iopub.status.idle":"2021-11-23T18:54:44.389813Z","shell.execute_reply.started":"2021-11-23T18:54:44.383989Z","shell.execute_reply":"2021-11-23T18:54:44.388772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile(OUT_TRAIN, 'w') as img_out:\n    for idx in tqdm(range(0,len(files))):\n        image = cv2.imread(f\"../input/sartorius-cell-instance-segmentation/train/{files[idx][0]}.png\")\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        temp_mask = image\n        M = temp_mask.shape[0]//2\n        N = temp_mask.shape[1]//2\n        tiles = [temp_mask[x:x+M,y:y+N] for x in range(0,temp_mask.shape[0],M) for y in range(0,temp_mask.shape[1],N)]\n        for j in range(4):\n            mask1 = tiles[j]\n            mask1 = cv2.imencode('.png',mask1)[1]\n            img_out.writestr(files[idx][0] + f'_{j}.png', mask1)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2021-11-23T18:54:44.391594Z","iopub.execute_input":"2021-11-23T18:54:44.391967Z","iopub.status.idle":"2021-11-23T18:55:03.423653Z","shell.execute_reply.started":"2021-11-23T18:54:44.39192Z","shell.execute_reply":"2021-11-23T18:55:03.422759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}