{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install classification-models-3D","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:46:56.023281Z","iopub.execute_input":"2021-10-06T16:46:56.023563Z","iopub.status.idle":"2021-10-06T16:47:02.048054Z","shell.execute_reply.started":"2021-10-06T16:46:56.023534Z","shell.execute_reply":"2021-10-06T16:47:02.047017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install efficientnet-3D keras_applications","metadata":{"execution":{"iopub.status.busy":"2021-08-28T07:43:15.389632Z","iopub.execute_input":"2021-08-28T07:43:15.389901Z","iopub.status.idle":"2021-08-28T07:43:21.309861Z","shell.execute_reply.started":"2021-08-28T07:43:15.389872Z","shell.execute_reply":"2021-08-28T07:43:21.308852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras_applications","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:02.049821Z","iopub.execute_input":"2021-10-06T16:47:02.050195Z","iopub.status.idle":"2021-10-06T16:47:08.850270Z","shell.execute_reply.started":"2021-10-06T16:47:02.050163Z","shell.execute_reply":"2021-10-06T16:47:08.849221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nfrom tensorflow import keras\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras.preprocessing import image\nimport cv2\nimport time\nimport glob\nimport os\nimport pandas\nfrom tensorflow.keras import layers\nfrom classification_models_3D.keras import Classifiers\ntf.random.set_seed(1)\nnp.random.seed(1)\n#random.seed(1)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:08.853857Z","iopub.execute_input":"2021-10-06T16:47:08.854148Z","iopub.status.idle":"2021-10-06T16:47:13.455788Z","shell.execute_reply.started":"2021-10-06T16:47:08.854119Z","shell.execute_reply":"2021-10-06T16:47:13.454980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_slices(df,base_dir): \n    all_paths = []\n    for i in list(df['folder_id']):\n        i = os.path.join(base_dir,i)\n        all_paths.append(len(glob.glob(i+'/flair/*')))\n    return all_paths\n\ndef split_train_test(slices_list,folders_list,label_list,split_ratio=0.1):\n    test_size = int(len(slices_list)*split_ratio)\n    test_slices_list = slices_list[:test_size]\n    test_folders_list = folders_list[:test_size]\n    test_label_list = label_list[:test_size]\n    train_slices_list = slices_list[test_size:]\n    train_folders_list = folders_list[test_size:]\n    train_label_list = label_list[test_size:]\n    return train_slices_list,train_folders_list,train_label_list,test_slices_list,test_folders_list,test_label_list","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:13.457286Z","iopub.execute_input":"2021-10-06T16:47:13.457595Z","iopub.status.idle":"2021-10-06T16:47:13.466677Z","shell.execute_reply.started":"2021-10-06T16:47:13.457562Z","shell.execute_reply":"2021-10-06T16:47:13.463606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/rsnasubmissionresult/result.csv',dtype='str')\nbase_dir = '../input/classify-tumor-best/DATATUMORONLY_TRAIN/train'\n#slices_list = np.array(get_all_slices(df,base_dir))","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:19.012771Z","iopub.execute_input":"2021-10-06T16:47:19.013137Z","iopub.status.idle":"2021-10-06T16:47:19.035302Z","shell.execute_reply.started":"2021-10-06T16:47:19.013105Z","shell.execute_reply":"2021-10-06T16:47:19.034561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df.iloc[:525,:]\ntest_df = df.iloc[526:,:]","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:20.446381Z","iopub.execute_input":"2021-10-06T16:47:20.446708Z","iopub.status.idle":"2021-10-06T16:47:20.453555Z","shell.execute_reply.started":"2021-10-06T16:47:20.446679Z","shell.execute_reply":"2021-10-06T16:47:20.450707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_slices_list = np.array(get_all_slices(train_df,base_dir))\ntest_slices_list = np.array(get_all_slices(test_df,base_dir))\n#slices_list = np.array(list(df['flair']))\ntrain_folders_list = np.array(list(train_df['folder_id']))\ntest_folders_list = np.array(list(test_df['folder_id']))\ntrain_label_list = np.array(list(train_df['MGMT_value']))\ntest_label_list = np.array(list(test_df['MGMT_value']))\nindexes = np.where((train_slices_list > 0 )&(train_slices_list < 50))\ntrain_slices_list = np.take(train_slices_list,indexes)[0]\ntrain_folders_list = np.take(train_folders_list,indexes)[0]\ntrain_label_list = np.take(train_label_list,indexes)[0]\nindexes = np.where((test_slices_list > 0 )&(test_slices_list < 50))\ntest_slices_list = np.take(test_slices_list,indexes)[0]\ntest_folders_list = np.take(test_folders_list,indexes)[0]\ntest_label_list = np.take(test_label_list,indexes)[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:21.989572Z","iopub.execute_input":"2021-10-06T16:47:21.989914Z","iopub.status.idle":"2021-10-06T16:47:25.055265Z","shell.execute_reply.started":"2021-10-06T16:47:21.989863Z","shell.execute_reply":"2021-10-06T16:47:25.054441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.read_csv('../input/rsnasubmissionresult/result.csv',dtype='str')\n# base_dir = '../input/classify-tumor-best/DATATUMORONLY_TRAIN/train'\n# slices_list = np.array(get_all_slices(df,base_dir))\n# #slices_list = np.array(list(df['flair']))\n# folders_list = np.array(list(df['folder_id']))\n# label_list = np.array(list(df['MGMT_value']))\n# indexes = np.where((slices_list > 0 )&(slices_list < 50))\n# slices_list = np.take(slices_list,indexes)[0]\n# folders_list = np.take(folders_list,indexes)[0]\n# label_list = np.take(label_list,indexes)[0]\n# shuffler = np.random.permutation(len(slices_list))\n# slices_list = slices_list[shuffler]\n# folders_list = folders_list[shuffler]\n# label_list = label_list[shuffler]\n# train_slices_list,train_folders_list,train_label_list,\\\n# test_slices_list,test_folders_list,test_label_list = split_train_test(slices_list,folders_list,label_list,split_ratio=0.1)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T05:32:26.881401Z","iopub.execute_input":"2021-09-15T05:32:26.881858Z","iopub.status.idle":"2021-09-15T05:32:32.655226Z","shell.execute_reply.started":"2021-09-15T05:32:26.881822Z","shell.execute_reply":"2021-09-15T05:32:32.654263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n    def __init__(self,slices_list,folders_list,label_list,width=256,height=256,batch_size=16,shuffle=True):\n        self.batch_size = batch_size\n        self.base_dir = '../input/classify-tumor-best/DATATUMORONLY_TRAIN/train'\n        self.width = width\n        self.crop_length = 224\n        self.height = height\n        self.tolerance = 5\n        self.shuffle = shuffle\n        self.intial_slices_list = slices_list\n        self.intial_folders_list = folders_list\n        self.intial_label_list = label_list\n        #print(len(self.slices_list))\n        self.on_epoch_end()\n    \n    def on_epoch_end(self):\n        print('epoch ended')\n        self.slices_list = self.intial_slices_list.copy()\n        self.folders_list = self.intial_folders_list.copy()\n        self.label_list = self.intial_label_list.copy()\n        if self.shuffle:\n            shuffler = np.random.permutation(len(self.slices_list))\n            self.slices_list = self.slices_list[shuffler]\n            self.folders_list = self.folders_list[shuffler]\n            self.label_list = self.label_list[shuffler]\n\n    def __len__(self):\n        return len(self.intial_slices_list)\n    \n    def __getitem__(self,user_index):\n        start =time.time()\n        index = self.slices_list[0]\n        #print(len(self.slices_list))\n        labels = []\n        indexes = np.where((self.slices_list >= index-self.tolerance) &(self.slices_list <= index+self.tolerance))\n        tol_slice= np.take(self.slices_list, indexes)[0]\n        tol_folder= np.take(self.folders_list, indexes)[0]\n        random_indexes = np.random.choice(indexes[0], size=min(self.batch_size,len(tol_folder)),replace=False)\n        random_folder = np.take(self.folders_list,random_indexes)\n        random_slices = np.take(self.slices_list,random_indexes)\n        random_labels = np.take(self.label_list,random_indexes)\n        self.folders_list = np.delete(self.folders_list,random_indexes)\n        self.slices_list = np.delete(self.slices_list,random_indexes)\n        self.label_list = np.delete(self.label_list,random_indexes)\n        #print(len(self.slices_list))\n        self.max_depth = random_slices.max()\n        #print(random_folder)\n        batch_x = self.__data_gen_batch(random_folder)\n        #for i in random_folder:\n        #    labels.append(int(self.label_list[np.where(self.folders_list == i)[0]][0]))\n        #print(labels)\n        return preprocess_input(batch_x),self.one_hot_encoder(random_labels.astype(np.int8))\n    \n    def one_hot_encoder(self,y):\n        b = np.zeros((len(y), 2))\n        b[np.arange(len(y)),y] = 1\n        return b\n    \n    def get_max_len(self,batch,min_depth=50):\n        max_len = 0\n        for patient_id in batch['folder_id']:\n            #print(os.path.join(self.base_dir,patient_id,'flair/*'))\n            length = len(glob.glob(os.path.join(self.base_dir,patient_id,'flair/*')))\n            if length > max_len:\n                max_len = length\n        if max_len < min_depth:\n            max_len = min_depth\n        return max_len\n\n    def __data_gen_image(self,folder_name):\n        flair_path = glob.glob(os.path.join(self.base_dir,folder_name,'flair/*'))\n        flair_path = sorted(flair_path,key=lambda x:x.split('-')[-1].split('.')[-2].zfill(3))\n        all_images = []\n        all_images = np.zeros(shape=(self.max_depth,self.height,self.height,1),dtype=np.float64)\n        for i,img_path in enumerate(flair_path):\n            img = image.load_img(img_path,target_size=(self.height,self.width),color_mode='grayscale')\n            img = image.img_to_array(img)\n            all_images[i,] = img\n        return np.transpose(all_images,(1,2,0,3))\n\n    def __data_gen_batch(self,folder_names):\n        batch_data = np.empty(shape=(len(folder_names),self.height,self.width,self.max_depth,1))\n        for i,patient_id in enumerate(folder_names):\n            batch_data[i,] = self.__data_gen_image(patient_id)\n        return batch_data\n    \n    def crop(self,image,crop_length=224):\n        img_height ,img_width = image.shape[:2]\n        start_y = (img_height - self.crop_length) // 2\n        start_x = (img_width - self.crop_length) // 2\n        cropped_image=image[start_y:(img_height - start_y), start_x:(img_width - start_x), :]\n        return cropped_image","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:25.354786Z","iopub.execute_input":"2021-10-06T16:47:25.355175Z","iopub.status.idle":"2021-10-06T16:47:25.380302Z","shell.execute_reply.started":"2021-10-06T16:47:25.355143Z","shell.execute_reply":"2021-10-06T16:47:25.379099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = DataGenerator(train_slices_list,train_folders_list,train_label_list,batch_size=5,height=224,width=224,shuffle=True)\ntest_datagen = DataGenerator(test_slices_list,test_folders_list,test_label_list,batch_size=1,height=224,width=224,shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:28.506570Z","iopub.execute_input":"2021-10-06T16:47:28.506909Z","iopub.status.idle":"2021-10-06T16:47:28.515921Z","shell.execute_reply.started":"2021-10-06T16:47:28.506858Z","shell.execute_reply":"2021-10-06T16:47:28.514724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for _ in range(3):\n#     print('new epoch')\n#     for i in range(len(test_datagen)-1):\n#         x,y = test_datagen[i]\n#         print(i,x.shape,len(test_datagen.slices_list))\n#     test_datagen.on_epoch_end()","metadata":{"execution":{"iopub.status.busy":"2021-09-16T17:09:27.567536Z","iopub.execute_input":"2021-09-16T17:09:27.567865Z","iopub.status.idle":"2021-09-16T17:09:31.741822Z","shell.execute_reply.started":"2021-09-16T17:09:27.567834Z","shell.execute_reply":"2021-09-16T17:09:31.740929Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# basemodel = efn.EfficientNetB0(input_shape=(256, 256, None, 1), weights=None)\n# x = layers.GlobalAveragePooling3D()(basemodel.output)\n# x = layers.Dense(units=128, activation=\"relu\")(x)\n# x = layers.Dropout(0.1)(x)\n# outputs = layers.Dense(units=2, activation=\"softmax\")(x)\n# # Define the model.\n# model = keras.Model(basemodel.input, outputs, name=\"eff3dcnn\")\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T16:52:36.075352Z","iopub.status.idle":"2021-08-28T16:52:36.076082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ResNet18, preprocess_input = Classifiers.get('resnet18')\nmodel = ResNet18(input_shape=(224, 224, None, 1), weights=None,include_top=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:30.391707Z","iopub.execute_input":"2021-10-06T16:47:30.392061Z","iopub.status.idle":"2021-10-06T16:47:32.759747Z","shell.execute_reply.started":"2021-10-06T16:47:30.392028Z","shell.execute_reply":"2021-10-06T16:47:32.758721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#x = layers.Dense(units=128, activation=\"relu\")(model.layers[-3].output)\n#x = layers.Dropout(0.1)(x)\noutputs = layers.Dense(units=2, activation=\"softmax\")(model.layers[-3].output)\n# Define the model.\nnew_model = keras.Model(model.input, outputs, name=\"resnet18_3d\")\nnew_model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:32.820312Z","iopub.execute_input":"2021-10-06T16:47:32.820548Z","iopub.status.idle":"2021-10-06T16:47:32.876934Z","shell.execute_reply.started":"2021-10-06T16:47:32.820524Z","shell.execute_reply":"2021-10-06T16:47:32.874869Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('models')\nos.makedirs('logs')","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:37.664483Z","iopub.execute_input":"2021-10-06T16:47:37.665037Z","iopub.status.idle":"2021-10-06T16:47:37.672916Z","shell.execute_reply.started":"2021-10-06T16:47:37.664995Z","shell.execute_reply":"2021-10-06T16:47:37.672062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_model.compile(\n    loss=\"categorical_crossentropy\",\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    metrics=[\"accuracy\"]\n)\ncheckpoint_cb = tf.keras.callbacks.ModelCheckpoint(\n    filepath=\"models/3d_image_classification.hdf5\", save_best_only=True,monitor=\"val_accuracy\",mode=\"max\",verbose=0)\nsummary = tf.keras.callbacks.TensorBoard(log_dir=\"./logs\",update_freq=1,histogram_freq=2)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:45.396823Z","iopub.execute_input":"2021-10-06T16:47:45.397230Z","iopub.status.idle":"2021-10-06T16:47:45.740968Z","shell.execute_reply.started":"2021-10-06T16:47:45.397186Z","shell.execute_reply":"2021-10-06T16:47:45.740088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_model.fit(\n    train_datagen,\n    steps_per_epoch=len(train_datagen)//5,\n    validation_data=test_datagen,\\\n    validation_steps=len(test_datagen)-2,\n    epochs=300,\n    verbose=1,\n    callbacks = [checkpoint_cb,summary]\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T16:47:48.222380Z","iopub.execute_input":"2021-10-06T16:47:48.222711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_model.save('best_50.hdf5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = DataGenerator(train_slices_list,train_folders_list,train_label_list,batch_size=5,height=224,width=224,shuffle=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_cnt = 0\nall_cnt = 0\nfor i in range(len(train_datagen)//5):\n    x,y = train_datagen[i]\n    y_pred = new_model.predict(x)\n    output = np.argmax(np.round_(y_pred,1),axis=1)==np.argmax(y,1)\n    true_cnt += sum(output)\n    all_cnt += len(output)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"true_cnt/all_cnt","metadata":{}}]}