{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob,cv2,os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:57.904466Z","iopub.execute_input":"2022-07-19T15:21:57.905902Z","iopub.status.idle":"2022-07-19T15:21:58.120014Z","shell.execute_reply.started":"2022-07-19T15:21:57.905758Z","shell.execute_reply":"2022-07-19T15:21:58.117667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Structuring","metadata":{}},{"cell_type":"code","source":"train_images = sorted(glob.glob('../input/hubmap-organ-segmentation/train_images/*'))\ntrain_annotations = sorted(glob.glob('../input/hubmap-organ-segmentation/train_annotations/*'))\n\nprint(f'Total Train Annotations : {len(train_annotations)}')\nprint(f'Total Train Images : {len(train_images)}')\n\ntrain_df = pd.DataFrame.from_dict({'id' : [os.path.splitext(os.path.basename(i))[0] for i in train_images],\n                                   'Train Images Name' : [os.path.basename(i) for i in train_images], \n                                   'Train Annotations Name' : [os.path.basename(i) for i in train_annotations],\n                                  'Train Images Path' : train_images, \n                                   'Train Annotations Path' : train_annotations})\n\ntrain_df['id'] = train_df['id'].astype('int64')\ntrain_df.sample(10)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T15:21:58.125921Z","iopub.execute_input":"2022-07-19T15:21:58.128524Z","iopub.status.idle":"2022-07-19T15:21:58.305859Z","shell.execute_reply.started":"2022-07-19T15:21:58.128474Z","shell.execute_reply":"2022-07-19T15:21:58.304886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')\nprint(train_csv.shape)\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:58.314504Z","iopub.execute_input":"2022-07-19T15:21:58.314861Z","iopub.status.idle":"2022-07-19T15:21:58.681503Z","shell.execute_reply.started":"2022-07-19T15:21:58.314827Z","shell.execute_reply":"2022-07-19T15:21:58.680322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_data = pd.merge(train_df, train_csv, on='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:58.685280Z","iopub.execute_input":"2022-07-19T15:21:58.685712Z","iopub.status.idle":"2022-07-19T15:21:58.706535Z","shell.execute_reply.started":"2022-07-19T15:21:58.685676Z","shell.execute_reply":"2022-07-19T15:21:58.705284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This DataFrame have all the details about training dataset\ntrain_df_data.sample(4)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:58.712415Z","iopub.execute_input":"2022-07-19T15:21:58.712898Z","iopub.status.idle":"2022-07-19T15:21:58.749313Z","shell.execute_reply.started":"2022-07-19T15:21:58.712858Z","shell.execute_reply":"2022-07-19T15:21:58.748232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Organ on Interest >> {train_df_data.organ.unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:58.751161Z","iopub.execute_input":"2022-07-19T15:21:58.752016Z","iopub.status.idle":"2022-07-19T15:21:58.760737Z","shell.execute_reply.started":"2022-07-19T15:21:58.751977Z","shell.execute_reply":"2022-07-19T15:21:58.759210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization of RLE Annotations ","metadata":{}},{"cell_type":"code","source":"def rle2mask(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    ref : https://www.kaggle.com/paulorzp/rle-functions-run-lenght-encode-decode\n    '''\n    s = mask_rle.split()\n    starts, lengths = [\n        np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])\n    ]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n    return img.reshape(shape).T\n\ndef visualize_annotations(dataframe,organ='any',resize=False,resize_percentage = 20):\n    if organ != 'any':\n        dataframe = dataframe[dataframe['organ'] == organ]\n    data = dataframe.sample(6)\n    plt.figure(figsize=(15,15))\n    for i in range(6):\n        plt.subplot(2,3,i+1)\n        msk = rle2mask(dataframe.iloc[i].rle,(dataframe.iloc[i].img_height,dataframe.iloc[i].img_width))\n        img = cv2.imread(dataframe.iloc[i]['Train Images Path'])\n        \n        if resize:\n            scale_percent = resize_percentage\n            width = int(img.shape[1] * scale_percent / 100)\n            height = int(img.shape[0] * scale_percent / 100)\n            dim = (width, height)\n            img = cv2.resize(img, dim, interpolation = cv2.INTER_AREA)\n            msk = cv2.resize(msk, dim, interpolation = cv2.INTER_AREA)\n        plt.imshow(img)\n        plt.imshow(msk,cmap='gray',alpha=0.2)\n        if resize:\n            plt.title(f'{dataframe.iloc[i].organ}, Data resized to : {dim}')\n        else:\n            plt.title(dataframe.iloc[i].organ)\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:58.764497Z","iopub.execute_input":"2022-07-19T15:21:58.765919Z","iopub.status.idle":"2022-07-19T15:21:58.786046Z","shell.execute_reply.started":"2022-07-19T15:21:58.765879Z","shell.execute_reply":"2022-07-19T15:21:58.784907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize (Random Selection of organs)\nvisualize_annotations(train_df_data,resize=True,resize_percentage=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:21:58.921175Z","iopub.execute_input":"2022-07-19T15:21:58.921671Z","iopub.status.idle":"2022-07-19T15:22:03.086934Z","shell.execute_reply.started":"2022-07-19T15:21:58.921640Z","shell.execute_reply":"2022-07-19T15:22:03.085938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization (Organ = Kidney)\nvisualize_annotations(train_df_data,organ='kidney')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:03.088515Z","iopub.execute_input":"2022-07-19T15:22:03.088826Z","iopub.status.idle":"2022-07-19T15:22:18.607275Z","shell.execute_reply.started":"2022-07-19T15:22:03.088796Z","shell.execute_reply":"2022-07-19T15:22:18.606293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling - Detect Organ","metadata":{}},{"cell_type":"code","source":"import tensorflow\nimport tensorflow as tf\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nimport numpy as np\nfrom tensorflow.keras.layers import Input\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:18.608727Z","iopub.execute_input":"2022-07-19T15:22:18.609516Z","iopub.status.idle":"2022-07-19T15:22:25.396581Z","shell.execute_reply.started":"2022-07-19T15:22:18.609479Z","shell.execute_reply":"2022-07-19T15:22:25.395328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def map_class(class_str):\n    _classes = ['prostate','spleen','lung','kidney','largeintestine']\n    _label = _classes.index(class_str)\n    return _label","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.403680Z","iopub.execute_input":"2022-07-19T15:22:25.406505Z","iopub.status.idle":"2022-07-19T15:22:25.416766Z","shell.execute_reply.started":"2022-07-19T15:22:25.406466Z","shell.execute_reply":"2022-07-19T15:22:25.415381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organ_detection_data = train_df_data.loc[:,['id',\n                                        'Train Images Path',\n                                       'organ']]\norgan_detection_data['Label'] = organ_detection_data['organ'].apply(map_class)\norgan_detection_data.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.418073Z","iopub.execute_input":"2022-07-19T15:22:25.423482Z","iopub.status.idle":"2022-07-19T15:22:25.448411Z","shell.execute_reply.started":"2022-07-19T15:22:25.423447Z","shell.execute_reply":"2022-07-19T15:22:25.447508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organ_detection_data_train, organ_detection_data_val = train_test_split(organ_detection_data,\n                                                                         test_size=0.2,\n                                                                         stratify=organ_detection_data['organ'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.452679Z","iopub.execute_input":"2022-07-19T15:22:25.455264Z","iopub.status.idle":"2022-07-19T15:22:25.466334Z","shell.execute_reply.started":"2022-07-19T15:22:25.455204Z","shell.execute_reply":"2022-07-19T15:22:25.464669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train Images Details --------')\norgan_detection_data_train.organ.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.468324Z","iopub.execute_input":"2022-07-19T15:22:25.469159Z","iopub.status.idle":"2022-07-19T15:22:25.483151Z","shell.execute_reply.started":"2022-07-19T15:22:25.469122Z","shell.execute_reply":"2022-07-19T15:22:25.482294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(organ_detection_data_train))\nprint(len(organ_detection_data_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.484057Z","iopub.execute_input":"2022-07-19T15:22:25.484400Z","iopub.status.idle":"2022-07-19T15:22:25.492038Z","shell.execute_reply.started":"2022-07-19T15:22:25.484367Z","shell.execute_reply":"2022-07-19T15:22:25.490933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Validation Images Details --------')\norgan_detection_data_val.organ.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.494037Z","iopub.execute_input":"2022-07-19T15:22:25.494670Z","iopub.status.idle":"2022-07-19T15:22:25.510677Z","shell.execute_reply.started":"2022-07-19T15:22:25.494634Z","shell.execute_reply":"2022-07-19T15:22:25.509787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Datagenerator","metadata":{}},{"cell_type":"code","source":"class DfDataGenerator(tf.keras.utils.Sequence):\n    \n    def __init__(self, \n                 batch_size,\n                 dataset_dataframe,\n                 dim=(224,224,3),\n                 n_classes=5,\n                 preprocess_input=None,\n                 shuffle=True):\n        \n        self.n_classes = n_classes\n        self.shuffle = shuffle\n        self.batch_size = batch_size\n        self.dataset_dataframe = dataset_dataframe\n        self.list_IDs = list(range(0,len(dataset_dataframe)))\n        self.dim = dim\n        self.on_epoch_end()\n        \n    def __len__(self):\n        # returns the number of batch\n        return len(self.dataset_dataframe) // self.batch_size\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size] # Generate indexes of the batch\n        list_IDs_temp = [self.list_IDs[k] for k in indexes] # Find list of IDs\n        X, y = self.__data_generation(list_IDs_temp) # Generate data\n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    def __data_generation(self, list_IDs_temp):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        \n        # Initialization\n        X = []\n        y = np.empty((self.batch_size), dtype=int)\n        \n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            \n            im_path = self.dataset_dataframe.iloc[i]['Train Images Path']\n            label = self.dataset_dataframe.iloc[i]['Label']\n            im = cv2.imread(im_path,cv2.COLOR_BGR2RGB)\n            im = cv2.resize(im, \n                            (self.dim[0],self.dim[1]), \n                            interpolation = cv2.INTER_AREA)\n            X.append(im)\n            y[i] = label\n            \n        X = np.stack(X)\n        if preprocess_input:\n            X = preprocess_input(X)\n        return X, tf.keras.utils.to_categorical(y, num_classes=self.n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.518194Z","iopub.execute_input":"2022-07-19T15:22:25.521074Z","iopub.status.idle":"2022-07-19T15:22:25.541774Z","shell.execute_reply.started":"2022-07-19T15:22:25.521027Z","shell.execute_reply":"2022-07-19T15:22:25.540545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_generator = DfDataGenerator(batch_size=7,\n                                   dataset_dataframe=organ_detection_data_train,\n                                   dim=(224,224,3),\n                                   n_classes=5,\n                                   preprocess_input=preprocess_input,\n                                   shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.546809Z","iopub.execute_input":"2022-07-19T15:22:25.549573Z","iopub.status.idle":"2022-07-19T15:22:25.557531Z","shell.execute_reply.started":"2022-07-19T15:22:25.549537Z","shell.execute_reply":"2022-07-19T15:22:25.555889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x,y = training_generator[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:25.561346Z","iopub.execute_input":"2022-07-19T15:22:25.562037Z","iopub.status.idle":"2022-07-19T15:22:28.523813Z","shell.execute_reply.started":"2022-07-19T15:22:25.562002Z","shell.execute_reply":"2022-07-19T15:22:28.522776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x.shape)\nprint(y.shape)\nim = x[0,:,:,:]\nprint(im.shape)\nplt.figure()\nplt.imshow(im)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:28.528776Z","iopub.execute_input":"2022-07-19T15:22:28.531206Z","iopub.status.idle":"2022-07-19T15:22:28.807223Z","shell.execute_reply.started":"2022-07-19T15:22:28.531164Z","shell.execute_reply":"2022-07-19T15:22:28.806180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    input_tensor = Input(shape=(224, 224, 3))\n    base_model = VGG16(weights='imagenet',input_tensor=input_tensor, include_top=False)\n    \n    for layer in base_model.layers:\n        layer.trainable = False\n        \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(1024, activation='relu')(x)\n    predictions = Dense(5, activation='softmax')(x)\n    model = Model(inputs=base_model.input, outputs=predictions)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:28.808711Z","iopub.execute_input":"2022-07-19T15:22:28.809055Z","iopub.status.idle":"2022-07-19T15:22:28.821476Z","shell.execute_reply.started":"2022-07-19T15:22:28.809020Z","shell.execute_reply":"2022-07-19T15:22:28.820305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_generator = DfDataGenerator(batch_size=4,\n                                   dataset_dataframe=organ_detection_data_train,\n                                   dim=(224,224,3),\n                                   n_classes=5,\n                                   preprocess_input=preprocess_input,\n                                   shuffle=True)\n\nval_generator = DfDataGenerator(batch_size=1,\n                                   dataset_dataframe=organ_detection_data_val,\n                                   dim=(224,224,3),\n                                   n_classes=5,\n                                   preprocess_input=preprocess_input,\n                                   shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:28.823362Z","iopub.execute_input":"2022-07-19T15:22:28.823703Z","iopub.status.idle":"2022-07-19T15:22:28.831570Z","shell.execute_reply.started":"2022-07-19T15:22:28.823669Z","shell.execute_reply":"2022-07-19T15:22:28.830628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m1 = create_model()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:22:28.833166Z","iopub.execute_input":"2022-07-19T15:22:28.833728Z","iopub.status.idle":"2022-07-19T15:22:32.263999Z","shell.execute_reply.started":"2022-07-19T15:22:28.833688Z","shell.execute_reply":"2022-07-19T15:22:32.262962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m1.compile(optimizer='rmsprop', \n           loss='categorical_crossentropy',\n           metrics=[tf.keras.metrics.CategoricalAccuracy()])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:23:58.061960Z","iopub.execute_input":"2022-07-19T15:23:58.062347Z","iopub.status.idle":"2022-07-19T15:23:58.078356Z","shell.execute_reply.started":"2022-07-19T15:23:58.062311Z","shell.execute_reply":"2022-07-19T15:23:58.077215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m1.fit_generator(generator=training_generator,\n                 validation_data=val_generator,\n                 epochs=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T15:24:01.128624Z","iopub.execute_input":"2022-07-19T15:24:01.129070Z","iopub.status.idle":"2022-07-19T15:25:14.548156Z","shell.execute_reply.started":"2022-07-19T15:24:01.129033Z","shell.execute_reply":"2022-07-19T15:25:14.547257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}