{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Import Libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport glob\nimport os\nimport torch\nfrom tqdm import tqdm\nimport tifffile\nfrom IPython.display import FileLink\nimport zipfile","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:30.346135Z","iopub.execute_input":"2022-08-28T04:21:30.346651Z","iopub.status.idle":"2022-08-28T04:21:32.963225Z","shell.execute_reply.started":"2022-08-28T04:21:30.346552Z","shell.execute_reply":"2022-08-28T04:21:32.961273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Helper Functions**","metadata":{}},{"cell_type":"code","source":"# tiff转img\ndef read_tiff(path): \n    img = tifffile.imread(path)\n    return img\n\n#read_tiff(\"/kaggle/input/hubmap-organ-segmentation/train_images/15329.tiff\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:32.967199Z","iopub.execute_input":"2022-08-28T04:21:32.968035Z","iopub.status.idle":"2022-08-28T04:21:32.978325Z","shell.execute_reply.started":"2022-08-28T04:21:32.967980Z","shell.execute_reply":"2022-08-28T04:21:32.974936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#读取mask\ndef rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction\n    \n    '''\n\n    # Split the string by space, then convert it into a integer array\n    s = np.array(mask_rle.split(), dtype=int)\n\n    # Every even value is the start, every odd value is the \"run\" length\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n\n    # The image image is actually flattened since RLE is a 1D \"run\"\n    if len(shape)==3:\n        h, w, d = shape\n        img = np.zeros((h * w, d), dtype=np.float32)\n    else:\n        h, w = shape\n        img = np.zeros((h * w,), dtype=np.float32)\n\n    # The color here is actually just any integer you want!\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n        \n    # Don't forget to change the image back to the original shape\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:32.980453Z","iopub.execute_input":"2022-08-28T04:21:32.981728Z","iopub.status.idle":"2022-08-28T04:21:32.995548Z","shell.execute_reply.started":"2022-08-28T04:21:32.981671Z","shell.execute_reply":"2022-08-28T04:21:32.994236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data**","metadata":{}},{"cell_type":"code","source":"DATA_DIR = '../input/hubmap-organ-segmentation'\ndf_tr = pd.read_csv(os.path.join(DATA_DIR, 'train.csv'))\nor_index = df_tr.organ.unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:32.997775Z","iopub.execute_input":"2022-08-28T04:21:32.998830Z","iopub.status.idle":"2022-08-28T04:21:33.359931Z","shell.execute_reply.started":"2022-08-28T04:21:32.998776Z","shell.execute_reply":"2022-08-28T04:21:33.357789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#获取.tiff文件路径\nTRAIN_IMAGES_DIR = \"/kaggle/input/../input/hubmap-organ-segmentation/train_images\"\nall_train_images = glob.glob(os.path.join(TRAIN_IMAGES_DIR, \"*.tiff\"), recursive=True)\n#print(all_train_images)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:33.362182Z","iopub.execute_input":"2022-08-28T04:21:33.363164Z","iopub.status.idle":"2022-08-28T04:21:33.404719Z","shell.execute_reply.started":"2022-08-28T04:21:33.363110Z","shell.execute_reply":"2022-08-28T04:21:33.403123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 创建文件夹\nbasedir_path = \"/kaggle/working/hubmap_compare\"\n    \nif not os.path.exists(basedir_path):\n    print(\"新建文件夹:\", basedir_path)\n    !mkdir /kaggle/working/hubmap_compare\n    !mkdir /kaggle/working/hubmap_compare/kidney\n    !mkdir /kaggle/working/hubmap_compare/kidney/train\n    !mkdir /kaggle/working/hubmap_compare/kidney/masks\n    !mkdir /kaggle/working/hubmap_compare/prostate\n    !mkdir /kaggle/working/hubmap_compare/prostate/train\n    !mkdir /kaggle/working/hubmap_compare/prostate/masks\n    !mkdir /kaggle/working/hubmap_compare/largeintestine\n    !mkdir /kaggle/working/hubmap_compare/largeintestine/train\n    !mkdir /kaggle/working/hubmap_compare/largeintestine/masks\n    !mkdir /kaggle/working/hubmap_compare/spleen\n    !mkdir /kaggle/working/hubmap_compare/spleen/train\n    !mkdir /kaggle/working/hubmap_compare/spleen/masks\n    !mkdir /kaggle/working/hubmap_compare/lung\n    !mkdir /kaggle/working/hubmap_compare/lung/train\n    !mkdir /kaggle/working/hubmap_compare/lung/masks\nelse:\n    print(\"已经存在该文件夹:\", basedir_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:33.410872Z","iopub.execute_input":"2022-08-28T04:21:33.411305Z","iopub.status.idle":"2022-08-28T04:21:51.780687Z","shell.execute_reply.started":"2022-08-28T04:21:33.411269Z","shell.execute_reply":"2022-08-28T04:21:51.778886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 加载.csv\nTR_CSV   = os.path.join(DATA_DIR, 'train.csv')\ntr_df = pd.read_csv(TR_CSV)[['id', 'rle', 'img_height', 'img_width', 'organ']] \ntr_df.head()\n#print(tr_df.shape)\ntr_df.loc[tr_df.organ == 'prostate']","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:51.783243Z","iopub.execute_input":"2022-08-28T04:21:51.783840Z","iopub.status.idle":"2022-08-28T04:21:52.004007Z","shell.execute_reply.started":"2022-08-28T04:21:51.783773Z","shell.execute_reply":"2022-08-28T04:21:52.003028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#存mask和png\nfor _, item in enumerate(or_index):\n    num = 0\n    print(item)\n    #存mask\n    mask_path = os.path.join(basedir_path, item, \"masks\")\n    train_path = os.path.join(basedir_path, item, \"train\")\n    maks_df = tr_df.loc[tr_df.organ == item]\n    if os.path.exists(mask_path) and len(os.listdir(mask_path)) == 0:\n        with tqdm(total = 5) as pbar:\n            for index, row in maks_df.iterrows():\n                if num < 5:\n                    ##mask\n                    #.csv转.png\n                    img = (np.stack([rle_decode(row['rle'], shape=(row['img_width'], row['img_height']))]*3, axis=-1)).astype(np.float32)\n                    #resize(512X512)\n                    img = cv2.resize(img, (512, 512), interpolation=cv2.INTER_LINEAR)\n                    cv2.imwrite(mask_path + \"/\" + str(row['id']) + '.png', img)\n                    print(num)\n                \n                    ##png\n                    for (i, train_img_path) in enumerate(all_train_images):\n                        idx = train_img_path[:-5].rsplit(\"/\", 1)[-1]\n                        if idx == str(row['id']):\n                            img_or = read_tiff(train_img_path).astype(np.int32)\n                            cv2.imwrite(train_path + \"/\" + idx + \".png\", \n                                img_or)  # tiff转img\n                \n                    pbar.update(1)\n                    num += 1\n                    if index == 1:\n                        plt.subplot(111)\n                        print(img.shape)\n                        plt.imshow((128 * img).astype(np.int32))\n                         ","metadata":{"execution":{"iopub.status.busy":"2022-08-28T04:21:52.005669Z","iopub.execute_input":"2022-08-28T04:21:52.006374Z","iopub.status.idle":"2022-08-28T04:22:29.028371Z","shell.execute_reply.started":"2022-08-28T04:21:52.006337Z","shell.execute_reply":"2022-08-28T04:22:29.026937Z"},"trusted":true},"execution_count":null,"outputs":[]}]}