{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob, sys, os\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\nimport warnings\nwarnings.filterwarnings(action = 'ignore')\nimport pandas as pd\nimport json\nimport tensorflow as tf\nimport tensorflow_io as tfio\nimport cv2\nfrom tensorflow import image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-18T05:11:38.993131Z","iopub.execute_input":"2022-08-18T05:11:38.993680Z","iopub.status.idle":"2022-08-18T05:11:46.189101Z","shell.execute_reply.started":"2022-08-18T05:11:38.993584Z","shell.execute_reply":"2022-08-18T05:11:46.187962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Load data","metadata":{}},{"cell_type":"code","source":"img_paths = sorted(glob.glob('../input/hubmap-organ-segmentation/train_images/*tiff'))\nprint(len(img_paths))\n\nplt.figure(figsize = (18,10))\nfor i in range(3) :\n    img = Image.open(img_paths[i])\n    plt.subplot(1,3,i+1)\n    plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:46.191379Z","iopub.execute_input":"2022-08-18T05:11:46.192175Z","iopub.status.idle":"2022-08-18T05:11:51.719046Z","shell.execute_reply.started":"2022-08-18T05:11:46.192129Z","shell.execute_reply":"2022-08-18T05:11:51.717968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')\nprint(df.shape)\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:51.720754Z","iopub.execute_input":"2022-08-18T05:11:51.721576Z","iopub.status.idle":"2022-08-18T05:11:52.061610Z","shell.execute_reply.started":"2022-08-18T05:11:51.721534Z","shell.execute_reply":"2022-08-18T05:11:52.060457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## organ(장기)\n* prostate : 전립선\n* spleen : 비장\n* lung : 폐\n* kidney : 신장\n* largeintestine : 대장\n\n## rle : run lengh encoding","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.064468Z","iopub.execute_input":"2022-08-18T05:11:52.065100Z","iopub.status.idle":"2022-08-18T05:11:52.088824Z","shell.execute_reply.started":"2022-08-18T05:11:52.065065Z","shell.execute_reply":"2022-08-18T05:11:52.087297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.Visualization","metadata":{}},{"cell_type":"code","source":"print(df['organ'].unique())\n\ncolumns = ['organ', 'sex', 'pixel_size']\nplt.figure(figsize = (15,6))\nfor i,col in enumerate(columns) :\n    plt.subplot(1,3,i+1)\n    sns.countplot(df[col])\n    plt.title(col, size = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.090497Z","iopub.execute_input":"2022-08-18T05:11:52.091301Z","iopub.status.idle":"2022-08-18T05:11:52.407588Z","shell.execute_reply.started":"2022-08-18T05:11:52.091254Z","shell.execute_reply":"2022-08-18T05:11:52.406468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df[df['sex'] == 'Male']\nm_count = []\nfor i in m['organ'].unique() :\n    a = m[m['organ'] == i]\n    m_count.append(len(a))\nf = df[df['sex'] == 'Female']\nf_count = [0]\nfor i in f['organ'].unique() :\n    a = f[f['organ'] == i]\n    f_count.append(len(a))\nprint(m.shape,f.shape)\n\nfig = plt.figure(figsize=(12, 8))\nidx = np.arange(1,10,2)\nplt.bar(idx+1.7, m_count, color = 'blue', width = 0.5, label = 'Male')\nplt.bar(idx+2.2, f_count, color = 'red', width = 0.5, label = 'Female')\nplt.xticks(idx+2, df['organ'].unique(), size = 15)\nplt.legend()\nplt.xlabel('Organ', size = 20)\nplt.ylabel('Counts', size = 20)\nplt.title('Organ distribution according to sex', size = 25)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.409128Z","iopub.execute_input":"2022-08-18T05:11:52.409472Z","iopub.status.idle":"2022-08-18T05:11:52.644600Z","shell.execute_reply.started":"2022-08-18T05:11:52.409441Z","shell.execute_reply":"2022-08-18T05:11:52.643453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(df['img_width'], df['img_height'], )","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.645946Z","iopub.execute_input":"2022-08-18T05:11:52.646298Z","iopub.status.idle":"2022-08-18T05:11:52.873221Z","shell.execute_reply.started":"2022-08-18T05:11:52.646263Z","shell.execute_reply":"2022-08-18T05:11:52.872151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# images shape\nshape = []\nfor i in range(len(df)):\n    s = (df['img_width'][i], df['img_height'][i])\n    shape.append(s)\nprint(set(shape))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.875161Z","iopub.execute_input":"2022-08-18T05:11:52.875863Z","iopub.status.idle":"2022-08-18T05:11:52.888887Z","shell.execute_reply.started":"2022-08-18T05:11:52.875819Z","shell.execute_reply":"2022-08-18T05:11:52.887249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['tissue_thickness'].unique(), df['data_source'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.890855Z","iopub.execute_input":"2022-08-18T05:11:52.891303Z","iopub.status.idle":"2022-08-18T05:11:52.900709Z","shell.execute_reply.started":"2022-08-18T05:11:52.891261Z","shell.execute_reply":"2022-08-18T05:11:52.899455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = glob.glob('../input/hubmap-organ-segmentation/test_images/*tiff')\nprint(test_path)\ntest_img = Image.open(test_path[0])\ntest_img = np.array(test_img)\nprint(test_img.shape)\nplt.figure(figsize = (15,5))\nplt.imshow(test_img)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:52.904127Z","iopub.execute_input":"2022-08-18T05:11:52.904600Z","iopub.status.idle":"2022-08-18T05:11:53.888804Z","shell.execute_reply.started":"2022-08-18T05:11:52.904568Z","shell.execute_reply":"2022-08-18T05:11:53.887717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/hubmap-organ-segmentation/test.csv')\nprint(df_test)\ndf_sub = pd.read_csv('../input/hubmap-organ-segmentation/sample_submission.csv')\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:53.890148Z","iopub.execute_input":"2022-08-18T05:11:53.890542Z","iopub.status.idle":"2022-08-18T05:11:53.913918Z","shell.execute_reply.started":"2022-08-18T05:11:53.890510Z","shell.execute_reply":"2022-08-18T05:11:53.912643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get train_annotations\nimport json\n\nann_paths = glob.glob('../input/hubmap-organ-segmentation/train_annotations/*json')\nanns = []\nfor path in ann_paths :\n    with open(path) as f:\n        ann = json.load(f)\n        anns.append(ann)\nanns = np.array(anns)\nprint(len(anns))\nanns.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:53.915648Z","iopub.execute_input":"2022-08-18T05:11:53.916084Z","iopub.status.idle":"2022-08-18T05:11:56.591778Z","shell.execute_reply.started":"2022-08-18T05:11:53.916043Z","shell.execute_reply":"2022-08-18T05:11:56.590716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(ann_paths[0]) as f :\n    a = json.load(f)\nlen(a)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:56.593199Z","iopub.execute_input":"2022-08-18T05:11:56.593659Z","iopub.status.idle":"2022-08-18T05:11:56.603843Z","shell.execute_reply.started":"2022-08-18T05:11:56.593618Z","shell.execute_reply":"2022-08-18T05:11:56.602657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 시작점과 끝나는점 리턴하는 함수 생성\ndef Rle_split(rle) :\n  n = np.array(rle.split(' '), dtype = int)\n  start = n[::2]\n  lengths = n[1::2]\n  end = start+lengths\n  return start,end\n\n# 백지 만들고 mask에 해당하는 부분 그리도록 하는 함수 생성\ndef Draw(shape, start, end, color) :\n  if len(shape) == 3 :\n    h,w,c = shape\n    white = np.zeros((h*w,c), dtype = np.float32)\n  else :\n    h,w = shape\n    white = np.zeros((h*w), dtype = np.float32)\n\n  for s,e in zip(start,end) :\n    white[s:e] = color\n  return white.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:56.605237Z","iopub.execute_input":"2022-08-18T05:11:56.605925Z","iopub.status.idle":"2022-08-18T05:11:56.614752Z","shell.execute_reply.started":"2022-08-18T05:11:56.605883Z","shell.execute_reply":"2022-08-18T05:11:56.613455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nfor i in range(6) :\n  rle = df['rle'][i]\n  start, end = Rle_split(rle)\n  a = Draw((3000,3000), start, end, 1)\n  plt.subplot(2,3,i+1)\n  # plt.xticks([]);plt.yticks([])\n  plt.axis(\"off\")\n  plt.imshow(a)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T05:11:56.616284Z","iopub.execute_input":"2022-08-18T05:11:56.616795Z","iopub.status.idle":"2022-08-18T05:12:02.989906Z","shell.execute_reply.started":"2022-08-18T05:11:56.616727Z","shell.execute_reply":"2022-08-18T05:12:02.988682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}