{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Simple Quick EDA**\n* One id have multiple encodings\n* 3 different types of cell types. Out of these 3, shsy5y is dominant in dataset.\n* All the images have fixed size.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom glob import glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-15T10:19:04.628866Z","iopub.execute_input":"2021-10-15T10:19:04.629136Z","iopub.status.idle":"2021-10-15T10:19:05.806850Z","shell.execute_reply.started":"2021-10-15T10:19:04.629108Z","shell.execute_reply":"2021-10-15T10:19:05.805887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"live_path='../input/sartorius-cell-instance-segmentation'\ntrain_path='../input/sartorius-cell-instance-segmentation/train'\ntest_path='../input/sartorius-cell-instance-segmentation/test'","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:05.808858Z","iopub.execute_input":"2021-10-15T10:19:05.809192Z","iopub.status.idle":"2021-10-15T10:19:05.814436Z","shell.execute_reply.started":"2021-10-15T10:19:05.809148Z","shell.execute_reply":"2021-10-15T10:19:05.813141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\nsample_sub=pd.read_csv('../input/sartorius-cell-instance-segmentation/sample_submission.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:05.816149Z","iopub.execute_input":"2021-10-15T10:19:05.816484Z","iopub.status.idle":"2021-10-15T10:19:06.506077Z","shell.execute_reply.started":"2021-10-15T10:19:05.816440Z","shell.execute_reply":"2021-10-15T10:19:06.505178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shape and length dataset\nprint('Train df shape: ',train_df.shape)\nprint('Total Unique Ids: ',train_df['id'].nunique())","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:06.507575Z","iopub.execute_input":"2021-10-15T10:19:06.507925Z","iopub.status.idle":"2021-10-15T10:19:06.528160Z","shell.execute_reply.started":"2021-10-15T10:19:06.507892Z","shell.execute_reply":"2021-10-15T10:19:06.527234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's check if all the images have same Dimensions\nprint('Total Unique Widths: ',train_df['width'].nunique())\nprint('Total Unique Heights: ',train_df['height'].nunique())","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:06.530395Z","iopub.execute_input":"2021-10-15T10:19:06.531235Z","iopub.status.idle":"2021-10-15T10:19:06.538646Z","shell.execute_reply.started":"2021-10-15T10:19:06.531187Z","shell.execute_reply":"2021-10-15T10:19:06.537893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Cell types\nprint('Types of cells: ',train_df['cell_type'].unique())\n\n#Distribution of cell types\nsns.displot(train_df['cell_type'])","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:06.539753Z","iopub.execute_input":"2021-10-15T10:19:06.539977Z","iopub.status.idle":"2021-10-15T10:19:07.159992Z","shell.execute_reply.started":"2021-10-15T10:19:06.539941Z","shell.execute_reply":"2021-10-15T10:19:07.159114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Now comes the main thing\n#let's take a look at some images and some masks then will take a loot at both","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:07.161254Z","iopub.execute_input":"2021-10-15T10:19:07.161500Z","iopub.status.idle":"2021-10-15T10:19:07.165916Z","shell.execute_reply.started":"2021-10-15T10:19:07.161469Z","shell.execute_reply":"2021-10-15T10:19:07.164866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_ids=train_df['id'].unique()\n\nr,c=2,2\nfig=plt.figure(figsize=(12,12))\nfor i in range(1,r*c+1):\n    img=cv2.imread(os.path.join(train_path,unique_ids[i]+'.png'))\n    fig.add_subplot(r,c,i)\n    plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:07.167159Z","iopub.execute_input":"2021-10-15T10:19:07.167534Z","iopub.status.idle":"2021-10-15T10:19:08.168495Z","shell.execute_reply.started":"2021-10-15T10:19:07.167496Z","shell.execute_reply":"2021-10-15T10:19:08.167755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:19:08.169766Z","iopub.execute_input":"2021-10-15T10:19:08.170149Z","iopub.status.idle":"2021-10-15T10:19:08.176733Z","shell.execute_reply.started":"2021-10-15T10:19:08.170097Z","shell.execute_reply":"2021-10-15T10:19:08.176048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets plot some images along with masks\nr,c=3,3\nfig=plt.figure(figsize=(18,18))\n\nfor i in range(1,r*c+1):\n    img=cv2.imread(os.path.join(train_path,unique_ids[i]+'.png'))\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    \n    fig.add_subplot(r,c,i)\n    \n    #Get annot\n    annots_=train_df['annotation'][train_df['id']==unique_ids[i]]\n    mask=np.zeros((502,704))\n    for ann in annots_.tolist():\n        mask+=rle_decode(ann,(502, 704))\n    \n    mask=np.clip(mask,0,1)\n    \n    #Get cell type\n    ctype=train_df['cell_type'][train_df['id']==unique_ids[i]].values[0]\n    \n    plt.imshow(img)\n    plt.imshow(mask,alpha=0.2,cmap='gray')\n    plt.title('Cell Type: '+ctype)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T10:35:03.974313Z","iopub.execute_input":"2021-10-15T10:35:03.974634Z","iopub.status.idle":"2021-10-15T10:35:06.855216Z","shell.execute_reply.started":"2021-10-15T10:35:03.974598Z","shell.execute_reply":"2021-10-15T10:35:06.851619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**What's next?**\n* Create a Classifier to classify images into 3 cell types\n* Create segmentation Model.","metadata":{}}]}