{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-04T13:24:05.843094Z","iopub.execute_input":"2021-12-04T13:24:05.843774Z","iopub.status.idle":"2021-12-04T13:24:05.871042Z","shell.execute_reply.started":"2021-12-04T13:24:05.843680Z","shell.execute_reply":"2021-12-04T13:24:05.870252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf1 = df.groupby('id').agg(list).reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:24:05.872610Z","iopub.execute_input":"2021-12-04T13:24:05.872851Z","iopub.status.idle":"2021-12-04T13:24:06.696239Z","shell.execute_reply.started":"2021-12-04T13:24:05.872824Z","shell.execute_reply":"2021-12-04T13:24:06.695473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[df1.id=='0030fd0e6378'].cell_type ","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:24:06.700355Z","iopub.execute_input":"2021-12-04T13:24:06.700999Z","iopub.status.idle":"2021-12-04T13:24:06.714831Z","shell.execute_reply.started":"2021-12-04T13:24:06.700955Z","shell.execute_reply":"2021-12-04T13:24:06.713992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[df.id=='ffdb3cc02eef'] ","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:24:06.716345Z","iopub.execute_input":"2021-12-04T13:24:06.716938Z","iopub.status.idle":"2021-12-04T13:24:06.722546Z","shell.execute_reply.started":"2021-12-04T13:24:06.716887Z","shell.execute_reply":"2021-12-04T13:24:06.721592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nfor col in df.columns[2:]:\n    df1[col] = df1[col].apply(\n        lambda x: np.unique(x)[0] if len(np.unique(x)) == 1 else np.unique(x)\n    )","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:24:06.724981Z","iopub.execute_input":"2021-12-04T13:24:06.725316Z","iopub.status.idle":"2021-12-04T13:24:07.177254Z","shell.execute_reply.started":"2021-12-04T13:24:06.725276Z","shell.execute_reply":"2021-12-04T13:24:07.176387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rles_to_mask(encs, shape):\n    \"\"\"\n    Decodes a rle.\n    #Thanks to Theovel\n    Args:\n        encs (list of str): Rles for each class.\n        shape (tuple [2]): Mask size.\n\n    Returns:\n        np array [shape]: Mask.\n    \"\"\"\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint)\n    mask_sum=np.zeros(len(encs))\n    for m, enc in enumerate(encs):\n         \n        if isinstance(enc, np.float) and np.isnan(enc):\n            continue\n        enc_split = enc.split()\n        for i in range(len(enc_split) // 2):\n            start = int(enc_split[2 * i]) - 1\n            length = int(enc_split[2 * i + 1])\n            img[start: start + length] = 1 + m\n        ##index=np.where(img==1+m)\n            \n        #print(img.sum())     \n        mask_sum[m]=img[np.where(img==1+m)].sum()\n        #break\n    return img.reshape(shape),mask_sum","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:24:07.178516Z","iopub.execute_input":"2021-12-04T13:24:07.178758Z","iopub.status.idle":"2021-12-04T13:24:07.187679Z","shell.execute_reply.started":"2021-12-04T13:24:07.178732Z","shell.execute_reply":"2021-12-04T13:24:07.186712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:24:07.189500Z","iopub.execute_input":"2021-12-04T13:24:07.189831Z","iopub.status.idle":"2021-12-04T13:24:07.433570Z","shell.execute_reply.started":"2021-12-04T13:24:07.189772Z","shell.execute_reply":"2021-12-04T13:24:07.432616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#i=cv2.imread('../input/sartorius-cell-instance-segmentation/train/0728b8f39241.png')\n#i=cv2.  equalizeHist(i.astype('int'))","metadata":{"execution":{"iopub.status.busy":"2021-11-20T09:53:01.532874Z","iopub.execute_input":"2021-11-20T09:53:01.533355Z","iopub.status.idle":"2021-11-20T09:53:01.569595Z","shell.execute_reply.started":"2021-11-20T09:53:01.533323Z","shell.execute_reply":"2021-11-20T09:53:01.568347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n#plt.imshow(i)","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:26:04.841463Z","iopub.execute_input":"2021-12-04T13:26:04.841939Z","iopub.status.idle":"2021-12-04T13:26:04.846268Z","shell.execute_reply.started":"2021-12-04T13:26:04.841891Z","shell.execute_reply":"2021-12-04T13:26:04.845313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df1['annotation']","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:25:10.522863Z","iopub.execute_input":"2021-12-04T13:25:10.523336Z","iopub.status.idle":"2021-12-04T13:25:10.536170Z","shell.execute_reply.started":"2021-12-04T13:25:10.523288Z","shell.execute_reply":"2021-12-04T13:25:10.535555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_class_dict={}\ndf1['mask_sum']=0\n\nfor i in range(df1.shape[0]):\n    \n\n    shape = df1[['height', 'width']].values[i]\n\n    mask=rles_to_mask(df1['annotation'][i],shape)\n    df1.loc[i,'mask_sum']=np.ceil(np.mean(mask[1] ))","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:25:24.905040Z","iopub.execute_input":"2021-12-04T13:25:24.905758Z","iopub.status.idle":"2021-12-04T13:25:45.592985Z","shell.execute_reply.started":"2021-12-04T13:25:24.905707Z","shell.execute_reply":"2021-12-04T13:25:45.592052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", 500)","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:37:02.530001Z","iopub.execute_input":"2021-12-04T13:37:02.530409Z","iopub.status.idle":"2021-12-04T13:37:02.535190Z","shell.execute_reply.started":"2021-12-04T13:37:02.530379Z","shell.execute_reply":"2021-12-04T13:37:02.534271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#class_dict={'shsy5y':10,'astro':20,'cort':30}\n#df1['class_mask_sum']=df1.cell_type.map(class_dict)\nplt.hist(df1.mask_sum,30)","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:33:40.799129Z","iopub.execute_input":"2021-12-04T13:33:40.799432Z","iopub.status.idle":"2021-12-04T13:33:41.203604Z","shell.execute_reply.started":"2021-12-04T13:33:40.799404Z","shell.execute_reply":"2021-12-04T13:33:41.202715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!  pip install category_encoders","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:42:11.419864Z","iopub.execute_input":"2021-12-04T13:42:11.420578Z","iopub.status.idle":"2021-12-04T13:42:21.768407Z","shell.execute_reply.started":"2021-12-04T13:42:11.420533Z","shell.execute_reply":"2021-12-04T13:42:21.767423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:39:47.190576Z","iopub.execute_input":"2021-12-04T13:39:47.190908Z","iopub.status.idle":"2021-12-04T13:39:47.214665Z","shell.execute_reply.started":"2021-12-04T13:39:47.190873Z","shell.execute_reply":"2021-12-04T13:39:47.213867Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df1[df1.cell_type.str.contains('ast')].mask_sum.hist()\n\n#df1['class_mask_sum_int']=df1.apply(lambda x: int((x['mask_sum'])%5),axis=1)\nimport category_encoders as ce\nfrom sklearn import preprocessing\ndf1['mask_sum_quantiles'] = pd.qcut(df1['mask_sum'], q=25, precision=0)\n\n#df1.mask_sum_quantiles.value_counts() \n\ndf1.groupby(['mask_sum_quantiles','cell_type']).id.count()\nle = preprocessing.LabelEncoder()\n\ndf1['mask_class']=le.fit_transform(df1['mask_sum_quantiles'])","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:47:03.742882Z","iopub.execute_input":"2021-12-04T13:47:03.743169Z","iopub.status.idle":"2021-12-04T13:47:03.753835Z","shell.execute_reply.started":"2021-12-04T13:47:03.743136Z","shell.execute_reply":"2021-12-04T13:47:03.752811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.groupby(['mask_class','cell_type']).id.count()","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:47:31.626837Z","iopub.execute_input":"2021-12-04T13:47:31.627350Z","iopub.status.idle":"2021-12-04T13:47:31.639643Z","shell.execute_reply.started":"2021-12-04T13:47:31.627309Z","shell.execute_reply":"2021-12-04T13:47:31.638714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.cell_type.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-17T11:10:55.683445Z","iopub.execute_input":"2021-11-17T11:10:55.683721Z","iopub.status.idle":"2021-11-17T11:10:55.692108Z","shell.execute_reply.started":"2021-11-17T11:10:55.683693Z","shell.execute_reply":"2021-11-17T11:10:55.691365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nskf=StratifiedKFold(n_splits=5, random_state=42, shuffle=True) \ndf1['fold']=-1\nfold=0\n#for trn_id,val_id in skf.split(df1.id.values,df1.class_mask_sum_int):\nfor trn_id,val_id in skf.split(df1.id.values,df1.mask_class):\n    #print(val_id)\n    df1.loc[val_id,'fold']=fold\n    fold=fold+1\n    \n","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:48:59.840493Z","iopub.execute_input":"2021-12-04T13:48:59.840838Z","iopub.status.idle":"2021-12-04T13:48:59.905560Z","shell.execute_reply.started":"2021-12-04T13:48:59.840769Z","shell.execute_reply":"2021-12-04T13:48:59.904425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-11-17T11:13:46.444092Z","iopub.execute_input":"2021-11-17T11:13:46.444371Z","iopub.status.idle":"2021-11-17T11:13:46.449222Z","shell.execute_reply.started":"2021-11-17T11:13:46.444341Z","shell.execute_reply":"2021-11-17T11:13:46.44833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.groupby(['fold','cell_type','mask_class']).id.count()","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:49:10.522912Z","iopub.execute_input":"2021-12-04T13:49:10.523595Z","iopub.status.idle":"2021-12-04T13:49:10.540653Z","shell.execute_reply.started":"2021-12-04T13:49:10.523559Z","shell.execute_reply":"2021-12-04T13:49:10.540059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.groupby(['fold','cell_type' ]).id.count()","metadata":{"execution":{"iopub.status.busy":"2021-12-04T13:50:42.363400Z","iopub.execute_input":"2021-12-04T13:50:42.363689Z","iopub.status.idle":"2021-12-04T13:50:42.374821Z","shell.execute_reply.started":"2021-12-04T13:50:42.363657Z","shell.execute_reply":"2021-12-04T13:50:42.373978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#mask.shape\n\n#mask[0][np.where(mask[0]==1)].sum()\n\ndf1.to_csv('sartorius_stratified_folds.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-16T07:54:00.324869Z","iopub.execute_input":"2021-11-16T07:54:00.325222Z","iopub.status.idle":"2021-11-16T07:54:00.33335Z","shell.execute_reply.started":"2021-11-16T07:54:00.325184Z","shell.execute_reply":"2021-11-16T07:54:00.33243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df1['class_length']=df1.cell_type.apply(lambda x :  np.array(x).shape[0])\n\n#Please let me know how does it correlates to LB\n ","metadata":{"execution":{"iopub.status.busy":"2021-11-17T11:19:42.485222Z","iopub.execute_input":"2021-11-17T11:19:42.486159Z","iopub.status.idle":"2021-11-17T11:19:42.493259Z","shell.execute_reply.started":"2021-11-17T11:19:42.48612Z","shell.execute_reply":"2021-11-17T11:19:42.49255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df1.class_length.unique()","metadata":{"execution":{"iopub.status.busy":"2021-11-13T10:16:26.111455Z","iopub.execute_input":"2021-11-13T10:16:26.111716Z","iopub.status.idle":"2021-11-13T10:16:26.121457Z","shell.execute_reply.started":"2021-11-13T10:16:26.111689Z","shell.execute_reply":"2021-11-13T10:16:26.120499Z"},"trusted":true},"execution_count":null,"outputs":[]}]}