{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch \nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom tqdm.notebook import tqdm\nimport albumentations as albu\nimport tensorflow as tf\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T16:56:53.815679Z","iopub.execute_input":"2022-07-06T16:56:53.816143Z","iopub.status.idle":"2022-07-06T16:57:06.810845Z","shell.execute_reply.started":"2022-07-06T16:56:53.816045Z","shell.execute_reply":"2022-07-06T16:57:06.809645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/code/iafoss/256x256-images/notebook","metadata":{}},{"cell_type":"code","source":"#functions to convert encoding to mask and mask to encoding\ndef enc2mask(encs, shape):\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for m,enc in enumerate(encs):\n        if isinstance(enc,np.float) and np.isnan(enc): continue\n        s = enc.split()\n        for i in range(len(s)//2):\n            start = int(s[2*i]) - 1\n            length = int(s[2*i+1])\n            img[start:start+length] = 1 + m\n    return img.reshape(shape).T\n\ndef rle_encode_less_memory(img):\n    pixels = img.T.flatten()\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T16:58:26.063291Z","iopub.execute_input":"2022-07-06T16:58:26.063686Z","iopub.status.idle":"2022-07-06T16:58:26.073244Z","shell.execute_reply.started":"2022-07-06T16:58:26.063655Z","shell.execute_reply":"2022-07-06T16:58:26.072091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PUT EVERYTHING INTO MEMORY RAM\nimages = read_images() #load images from disk into memory\nmasks = decode_RLE() #0's with 1's where mask is\nLIMIT = 10\n\n# DATA LOADER BELOW\nfor j in range(LIMIT):\n    k = pick_image() #returns a number from 0 to len(images)-1\n    a,b,c,d = pick_random_crop(masks[k]) #returns random bounds\n    mask = masks[k][a:b,c:d]\n    if np.sum( mask )>0: break\nimage = images[k][a:b,c:d, :]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"../input/hubmap-organ-segmentation/train_images/10044.tiff","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train= pd.read_csv('../input/hubmap-organ-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T16:57:37.298571Z","iopub.execute_input":"2022-07-06T16:57:37.299143Z","iopub.status.idle":"2022-07-06T16:57:37.655161Z","shell.execute_reply.started":"2022-07-06T16:57:37.299104Z","shell.execute_reply":"2022-07-06T16:57:37.653937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T16:57:49.166611Z","iopub.execute_input":"2022-07-06T16:57:49.167411Z","iopub.status.idle":"2022-07-06T16:57:49.191857Z","shell.execute_reply.started":"2022-07-06T16:57:49.167365Z","shell.execute_reply":"2022-07-06T16:57:49.190765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Strategy:\n-Create a tile function that will create patches for the model to train on\n","metadata":{}},{"cell_type":"code","source":"train_images=[]\ntrain_masks=[]\n\ndef load_image(df, img_list, mask_list, )\nfor k in range(len(train.index)):\n    name= train.iloc[k,0]\n    img= np.squeeze(tiff.imread('../input/hubmap-organ-segmentation/train_images'+name+'.tiff'))\n    if img.shape[0]==3: img=np.transpose(img,[1,2,0])    #want to change shape from CxHxW to HxWxC\n    train_images.append(img)\n    rle= train['rle'][k]\n    mask= enc2mask(rle, shape=(img.shape[1],img.shape[0])) #this fuction needs the width first instead of height\n    train_masks.append(mask)\n    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import train_test_split","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, y_train, X_val, y_val= train_test_split(train_images, train_masks,test_size= 0.2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/competitions/hubmap-kidney-segmentation/discussion/238308","metadata":{}},{"cell_type":"markdown","source":"Here is a custom dataloader, that inherits from tf.keras.utils.Sequence. The images in the dataset are very large (3000x3000), so it is important to decompose these images into smaller patches/tiles ","metadata":{}},{"cell_type":"code","source":"class hubmap(tf.keras.utils.Sequence):\n    def __init__(self,imgs, masks, batch_size=24, shuffle=False, augment=False, crops=128, size, size2, shrink):\n        self.imgs = imgs\n        self.masks = masks\n        self.batch_size = batch_size\n        self.crops = crops\n        self.size = size\n        self.size2 = size2\n        self.shrink = shrink\n        self.shuffle = shuffle\n        self.augment = augment\n        self.on_epoch_end()\n        self.total_crops=self.crops*len(self.imgs)\n        \n    def __len__(self):\n        ct= int(np.ceil(self.total_crops/self.batch_size)) #number of batches per epoch, (total no of crops)/no. of crops per batches\n        return ct\n    \n    def __getitem__(self, index):\n        #Generates 1 batch of data (batch specified by index)\n        indexes= self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X,y= self.__data_generation(indexes)\n        if self.augment:\n            X2= np.zeros((len(indexes), self.size2,self.size2,3), dtype='float32') #np array to populate of dims: batchsize x size2 x size2 x 3 (3 for no. of channels)\n            y2= np.zeros((len(indexes, self.size2, self.size2,1), dtype='float32')) #mask with same dimensions except for channels which is just equal to the no of classes (1)\n            X2,y2= self.__augment_batch(X,y)\n        else:\n            X2=X\n            y2=y\n    def on_epoch_end(self): #updates indexes after each epoch\n        self.indexes= np.arange(self.total_crops)\n        if self.shuffle: np.random.shuffle(self.indexes)\n            \n    def __data_generation(self,indexes): #Generates data containing batch size samples\n        \n        X=np.zeros((len(indexes), self.size2,self.size2, 3), dtype='float32')\n        y=np.zeros((len(indexes),self.size2,self.size2,1), dtype='float32')\n\n        for k in range(len(indexes)): #for k in range of batch size\n            i = np.random.randint(0, len(self.imgs))\n            mask= self.masks[i]\n            img=self.imgs[i]\n            \n            sm=0\n            ct=0\n            while (sm==0)&(ct<25): #will repeat until sum>0, ensures that the mask contains at least 1 segmentation label\n                a= np.random.randint(0, img.shape[0]-self.size2*self.shrink) #Subtraction to ensure that when we add selfsize*shrink, it will stay within the dims \n                b= np.random.randint(0, img.shape[1]-self.size2*self.shrink)\n                sm=np.sum(mask[a:a+self.size2*self.shrink, b:b+self.size2*self.shrink])\n                ct+=1\n                \n            X[k,]= img[a:a+self.size2*self.shrink,b:b+self.size2*self.shrink,][::self.shrink, ::self.shrink]/255    #a numpy trick which reduces the image by the amount specified by shrink\n            y[k,:,:,0]= mask[a:a+self.size2*self.shrink,b:b+self.size2*self.shrink][::self.shrink,::self.shrink] #for class0\n            return X,y\n        \n    def __transform(self, img, mask):\n        composition = albu.Compose([\n            albu.HorizontalFlip(p=0.5),\n            albu.VerticalFlip(p=0.5),\n            albu.ShiftScaleRotate(rotate_limit=25, scale_limit=0.15, shift_limit=0, p=0.75),\n            albu.CoarseDropout(max_holes=16, max_height=64 ,max_width=64 ,p=0.5),\n            albu.ColorJitter(brightness=0.25, contrast=0.25, saturation=0.25, hue=0.25, p=0.75),\n            albu.GridDistortion(num_steps=5, distort_limit=0.3, interpolation=1, p=0.5),\n            albu.RandomCrop(height=self.size, width=self.size, p=1.0)    #Selects a random crop of the image with the specified size\n        ])\n        tmp = composition(img, mask)\n        return tmp['image'], tmp['mask']\n    \n    def __augment_batch(self, img_batch, mask_batch):\n        img_batch2 = np.zeros((len(img_batch),self.size,self.size,3))\n        mask_batch2 = np.zeros((len(mask_batch),self.size,self.size,1))\n        for i in range(img_batch.shape[0]):\n            img_batch2[i, ],mask_batch2[i, ] = self.__random_transform(img_batch[i, ],mask_batch[i, ])\n        return img_batch2,mask_batch2\n            ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tifffile as tiff","metadata":{"execution":{"iopub.status.busy":"2022-07-06T16:59:55.244859Z","iopub.execute_input":"2022-07-06T16:59:55.245271Z","iopub.status.idle":"2022-07-06T16:59:55.405680Z","shell.execute_reply.started":"2022-07-06T16:59:55.245239Z","shell.execute_reply":"2022-07-06T16:59:55.404479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below I have demonstrated on just 1 example what the above class will do. The image and mask have been cropped to the specified dimensions, however the mask did ot contain a segmentation label. For the code above however we have ensured that this will not be the case.","metadata":{}},{"cell_type":"code","source":"img= np.squeeze(tiff.imread('../input/hubmap-organ-segmentation/train_images/10044.tiff'))\nif img.shape[0]==3: img=np.transpose(img,[1,2,0])\nrle= train['rle'][0]\nmask= enc2mask(rle, shape=(img.shape[1],img.shape[0]))\na= np.random.randint(0, img.shape[0]-512*3) #Subtraction to ensure that when we add selfsize*shrink, it will stay within the dims \nb= np.random.randint(0, img.shape[1]-512*3)\nimg= img[a:a+512*3,b:b+512*3,][::3, ::3]/255 \nmask = mask[a:a+512*3,b:b+512*3][::3, ::3]       \ncropped_image = tf.image.random_crop(img, size=[224, 224, 3])   \ncropped_mask= tf.image.random_crop(mask, size=[224,224])\nplt.subplot(1,2,1)\nplt.imshow(cropped_image)\nplt.subplot(1,2,2)\nplt.imshow(cropped_mask)\n   ","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:12:53.893490Z","iopub.execute_input":"2022-07-06T17:12:53.893889Z","iopub.status.idle":"2022-07-06T17:12:54.313527Z","shell.execute_reply.started":"2022-07-06T17:12:53.893856Z","shell.execute_reply":"2022-07-06T17:12:54.312380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:10:06.242562Z","iopub.execute_input":"2022-07-06T17:10:06.243274Z","iopub.status.idle":"2022-07-06T17:10:06.249030Z","shell.execute_reply.started":"2022-07-06T17:10:06.243235Z","shell.execute_reply":"2022-07-06T17:10:06.248121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds= hubmap(X_train, y_train, augment=True,size=512,size2=256,shrink=3)\nval_ds= hubmap(X_val, y_val, augment=True, size=512, size2=256, shrink=3)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import segmentation_models as sm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Historically, UNet has been the architecture used for semantic segmentation. \n.fit_generator is used when either we have a huge dataset to fit into our memory or when data augmentation needs to be applied.","metadata":{}},{"cell_type":"code","source":"BACKBONE= 'resnet34'\npreprocess_input=sm.get_preprocessing\n\nmetrics = [sm.metrics.IOUScore(threshold=0.5), sm.metrics.FScore(threshold=0.5)]\nmodel = sm.Unet(BACKBONE,input_shape=(256,256,3),encoder_weights='imagenet' classes=1, activation='sigmoid')\nmodel.fit_generator(train_ds, validation_data=(val_ds), steps_per_epoch=len(train_ds),validation_data=val_ds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://github.com/qubvel/segmentation_models/blob/master/examples/binary%20segmentation%20(camvid).ipynb","metadata":{}}]}