{"cells":[{"metadata":{},"cell_type":"markdown","source":"# OutPut is RLE encoding for every images that has been generated in \n[This image (cropped images of object that we have to find)](http://)"},{"metadata":{"_uuid":"9e6d9e16-860b-44ba-a356-8eaabadb4dac","_cell_guid":"2bb05e9b-d0e2-48ec-9903-2cf19d348723","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport zipfile \nimport matplotlib.pyplot as plt\nimport matplotlib\nimport seaborn as sns\nimport tifffile\nimport cv2 as cv\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n'''\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"463fe285-2e2d-443c-9472-7e53e67c95c1","_cell_guid":"9929aa71-f16b-482a-86ce-760576050cbf","trusted":true},"cell_type":"markdown","source":"* cropping mask image\n* get there encoding\n* **This is process of image segmentation**"},{"metadata":{"_uuid":"64e2bfef-dfd1-4325-963e-6e33f2e24a33","_cell_guid":"258e1bfd-1c01-45cb-9c9d-035325b2a11c","trusted":true},"cell_type":"code","source":"\"\"\"\ntrain_csv = pd.read_csv(\"../input/hubmap-kidney-segmentation/train.csv\")\nlist_ID = list()\nfor i in range(len(train_csv)):\n    list_ID.append(train_csv.id[i])\nprint(list_ID)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1e43ff8d-08da-4dbb-980a-319ffac29ca5","_cell_guid":"65f1aa35-a8df-47e7-b6a0-8423a4d0c657","trusted":true},"cell_type":"code","source":"\"\"\"\n#get an image \nimport tifffile\ndef get_image(ID):\n    path = \"../input/hubmap-kidney-segmentation/train/\"+ID+\".tiff\"\n    return tifffile.imread(path)\nimg = get_image(list_ID[3])\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6be6d7f1-c6a1-4736-8959-6a9e64bcafb0","_cell_guid":"48719797-1cef-4952-b6ef-50e9d7d1449b","trusted":true},"cell_type":"code","source":"\"\"\"\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(10,16))\nplt.imshow(img)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b45d6816-d15c-4997-8fc7-8cdf939d9de1","_cell_guid":"3d15a35e-c782-462c-99f8-b2106f083c3b","trusted":true},"cell_type":"code","source":"\"\"\"\n# get the mask image or rletomask\nID = list_ID[3]\ndef rle2mask(mask_rle,shape):\n    s = mask_rle.split()\n    starts, lengths = [\n        np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])\n    ]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n    return img.reshape(shape).T\nmask = rle2mask(train_csv[train_csv[\"id\"]==ID][\"encoding\"].values[0],(img.shape[1],img.shape[0]),)\nplt.figure(figsize=(10,16))\nplt.imshow(mask)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7d06b7de-6de0-4531-aafd-a3f0ae4229d8","_cell_guid":"7c96b802-b04b-46ed-b3e5-8a1c1f6287fb","trusted":true},"cell_type":"code","source":"\"\"\"\ndef Coor_of_Glomeruli(ID,n):\n    df = pd.read_json(\"../input/hubmap-kidney-segmentation/train/\"+ID+\".json\")\n    return np.array(df.geometry[n][\"coordinates\"])[0,:,:],n\n# Got the corrdinates\ndef find_box(coor_array,h,w):\n    L_x = coor_array[0][0]\n    S_x = L_x\n    L_y = coor_array[0][1]\n    S_y = L_y\n    for i in range(len(coor_array)):\n        if L_x < coor_array[i][0]:\n            L_x = coor_array[i][0]\n    for i in range(len(coor_array)):\n        if S_x > coor_array[i][0]:\n            S_x = coor_array[i][0]\n    for i in range(len(coor_array)):\n        if L_y < coor_array[i][1]:\n            L_y = coor_array[i][1]\n    for i in range(len(coor_array)):\n        if S_y > coor_array[i][1]:\n            S_y = coor_arraty[i][1]\n    return L_x,S_x,L_y,S_y\n# got the box\ndef save_image(sub_image,image_id,no_of_glomeruli):\n    path = '../working/'+image_id+\"/\"+str(no_of_glomeruli)+'.png'\n    #ther many glomeruli in image so no_of_glomeruli represent that \n    cv.imwrite(path,sub_image)\n    \ndef total_glomeruli(ID):\n    df = pd.read_json(\"../input/hubmap-kidney-segmentation/train/\"+ID+\".json\")\n    return len(df.geometry)\n\n\n\n    \nfor i in range(len(train_csv.id)):\n    image_id = list_ID[3]\n    #get the coordinates array of each glomeruli\n    image = mask\n    Total_Glomeruli = total_glomeruli(image_id)\n    w,h = image.shape[:2]\n    #path = os.path.join(\"../working\",image_id)\n    #os.mkdir(path,0o666)\n    for j in range(1):#Total_Glomeruli):\n        coor_array,no_of_Glomeruli = Coor_of_Glomeruli(image_id,j)\n        #get the box\n        #print(\"shape of array\",coor_array.shape)\n        L_x,S_x,L_y,S_y = find_box(coor_array,h,w)\n        #crop the image\n        #print(\"box coordinate\",L_x,S_x,L_y,S_y)\n        #plt.figure(figsize=(10,16))\n        if S_x < w and S_y < h and L_x < w and L_y < h:\n            sub_image = image[S_y:L_y,S_x:L_x]    \n        # save the cropped image\n        #print('sub image shape',sub_image.shape)\n        #print('image shape',image.shape)\n        #print('image ID',image_id)\n        plt.imshow(sub_image)\n        #save_image(sub_image,image_id,no_of_Glomeruli)\n        #start with new image\n    print(\"done Id \",image_id)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6ae30bdf-6691-4c91-a145-8c969786a8da","_cell_guid":"b8b6edf0-6b0b-4b51-93c6-5733c6c5fb17","trusted":true},"cell_type":"code","source":"\"\"\"\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    This simplified method requires first and last pixel to be zero\n    '''\n    pixels = img.T.flatten()\n    \n    # This simplified method requires first and last pixel to be zero\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    \n    return ' '.join(str(x) for x in runs)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d33c425b-6c27-49a2-bc0e-3f61f44b1695","_cell_guid":"bde0b9fc-63c4-4f93-b22d-a0f4dbce7ae9","trusted":true},"cell_type":"code","source":"\"\"\"\n\nprint(rle_encode(mask))\n\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4393afeb-73db-452d-9eb3-61628cb5609a","_cell_guid":"53605d0c-5b9b-4072-be0c-8ddc1125c3bf","trusted":true},"cell_type":"code","source":"\"\"\"\n\nprint(train_csv.encoding[3])\n\n\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4a0d91ec-74c9-44d1-b38c-ddafc7e591cb","_cell_guid":"118fc37b-c9bd-422f-ac91-11b7ce15478c","trusted":true},"cell_type":"code","source":"# a cell that use all the above function and variable to create a the image cropping\ndef get_id(n):\n    return df_train.id[n]\n\n\ndef Coor_of_Glomeruli(ID,n):\n    df = pd.read_json(\"../input/hubmap-kidney-segmentation/train/\"+ID+\".json\")\n    return np.array(df.geometry[n][\"coordinates\"])[0,:,:],n\n\n\ndef get_image(ID):\n    return tifffile.imread(\"../input/hubmap-kidney-segmentation/train/\"+ID+\".tiff\")\n\n\ndef find_box(coor_array,h,w):\n    L_x = coor_array[0][0]\n    S_x = L_x\n    L_y = coor_array[0][1]\n    S_y = L_y\n    for i in range(len(coor_array)):\n        if L_x < coor_array[i][0]:\n            L_x = coor_array[i][0]\n    for i in range(len(coor_array)):\n        if S_x > coor_array[i][0]:\n            S_x = coor_array[i][0]\n    for i in range(len(coor_array)):\n        if L_y < coor_array[i][1]:\n            L_y = coor_array[i][1]\n    for i in range(len(coor_array)):\n        if S_y > coor_array[i][1]:\n            S_y = coor_arraty[i][1]\n    return L_x,S_x,L_y,S_y\n\n\ndef mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n \n\n\n\ndef save_image(sub_image,image_id,no_of_glomeruli):\n    path = '../working/'+image_id+\"_\"+str(no_of_glomeruli)+'.png'\n    #ther many glomeruli in image so no_of_glomeruli represent that \n    cv.imwrite(path,sub_image)\n\n    \ndef total_glomeruli(ID):\n    df = pd.read_json(\"../input/hubmap-kidney-segmentation/train/\"+ID+\".json\")\n    return len(df.geometry)\n\n\n\ndef rle2mask(mask_rle,shape):\n    s = mask_rle.split()\n    starts, lengths = [\n    np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])\n    ]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n    return img.reshape(shape).T\n\n########################\n########################\ndf_train = pd.read_csv(\"../input/hubmap-kidney-segmentation/train.csv\")\ntrain_data = pd.DataFrame(data=None,columns=['file_name','rle',],index=None) # creating a empyt data\n\n\npath = os.path.join(\"../working/\")\n#os.mkdir(path,)\n########################\n########################\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# getting rle \n    ## cropping the mask\nfor i in range(len(df_train.id)):\n    image_id = get_id(i) #got the image id \n    #get the coordinates array of each glomeruli\n    image = tifffile.imread(\"../input/hubmap-kidney-segmentation/train/\"+image_id+\".tiff\")\n    image = rle2mask(df_train[df_train[\"id\"]==image_id][\"encoding\"].values[0],(image.shape[1],image.shape[0]),)\n    Total_Glomeruli = total_glomeruli(image_id)\n    w,h = image.shape[:2]\n    for j in range(Total_Glomeruli):\n        coor_array,no_of_glomeruli = Coor_of_Glomeruli(image_id,j)\n        #get the box\n        L_x,S_x,L_y,S_y = find_box(coor_array,h,w)\n        #crop the mask\n        if S_x < w and S_y < h and L_x < w and L_y < h:\n            sub_image = image[S_y:L_y,S_x:L_x]    \n        # now get rle sub image\n        rle = mask2rle(sub_image)\n        # now name of image\n        name_of_image = image_id+\"/_\"+str(no_of_glomeruli)+'.png'\n        # write the rle image name in dataframe\n        new_row = {'file_name':name_of_image,'rle':rle,}\n        train_data=train_data.append(new_row,ignore_index=True)\n                \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('creating data_csv........')\ntrain_data.to_csv(path+'train_data.csv',index=False)\nprint('csv createad')\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}