{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport glob\nimport random\nimport collections\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:18:53.469983Z","iopub.execute_input":"2021-12-12T00:18:53.470511Z","iopub.status.idle":"2021-12-12T00:18:54.944203Z","shell.execute_reply.started":"2021-12-12T00:18:53.470473Z","shell.execute_reply":"2021-12-12T00:18:54.943129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:00.943533Z","iopub.execute_input":"2021-12-12T00:19:00.943912Z","iopub.status.idle":"2021-12-12T00:19:00.983119Z","shell.execute_reply.started":"2021-12-12T00:19:00.943883Z","shell.execute_reply":"2021-12-12T00:19:00.982111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#seperate dataset into two types by 'MGMT_value'\ntrain_df_0 = train_df[train_df['MGMT_value']==0]\ntrain_df_1 = train_df[train_df['MGMT_value']==1]","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:07.853364Z","iopub.execute_input":"2021-12-12T00:19:07.853775Z","iopub.status.idle":"2021-12-12T00:19:07.881185Z","shell.execute_reply.started":"2021-12-12T00:19:07.853739Z","shell.execute_reply":"2021-12-12T00:19:07.879932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load .dcm image as numpy array\ndef load_dicom(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:14.000354Z","iopub.execute_input":"2021-12-12T00:19:14.001091Z","iopub.status.idle":"2021-12-12T00:19:14.008304Z","shell.execute_reply.started":"2021-12-12T00:19:14.001046Z","shell.execute_reply":"2021-12-12T00:19:14.00751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"package_path = \"../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master/\"\nimport sys \nsys.path.append(package_path)\n\nimport time\n\nimport torch\nfrom torch import nn\nfrom torch.utils import data as torch_data\nfrom sklearn import model_selection as sk_model_selection\nfrom torch.nn import functional as torch_functional\nimport efficientnet_pytorch\n\nfrom sklearn.model_selection import StratifiedKFold\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:17.634044Z","iopub.execute_input":"2021-12-12T00:19:17.634409Z","iopub.status.idle":"2021-12-12T00:19:19.13789Z","shell.execute_reply.started":"2021-12-12T00:19:17.63438Z","shell.execute_reply":"2021-12-12T00:19:19.136128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed_all(seed)\n        torch.backends.cudnn.deterministic = True\n\n\nset_seed(42)","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:22.499439Z","iopub.execute_input":"2021-12-12T00:19:22.499805Z","iopub.status.idle":"2021-12-12T00:19:22.508276Z","shell.execute_reply.started":"2021-12-12T00:19:22.499775Z","shell.execute_reply":"2021-12-12T00:19:22.507188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prepare training and validation dataset\ndf = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\ndf_train, df_valid = sk_model_selection.train_test_split(\n    df, \n    test_size=0.2, \n    random_state=42, \n    stratify=train_df[\"MGMT_value\"],\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:28.335456Z","iopub.execute_input":"2021-12-12T00:19:28.335993Z","iopub.status.idle":"2021-12-12T00:19:28.351331Z","shell.execute_reply.started":"2021-12-12T00:19:28.335961Z","shell.execute_reply":"2021-12-12T00:19:28.350189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#seperate training dataset into two types by 'MGMT_value'\ndf_train_0 = df_train[df_train['MGMT_value']==0]\ndf_train_1 = df_train[df_train['MGMT_value']==1]\n\n#seperate val dataset into two types by 'MGMT_value'\ndf_valid_0 = df_valid[df_valid['MGMT_value']==0]\ndf_valid_1 = df_valid[df_valid['MGMT_value']==1]","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:33.557147Z","iopub.execute_input":"2021-12-12T00:19:33.557544Z","iopub.status.idle":"2021-12-12T00:19:33.565867Z","shell.execute_reply.started":"2021-12-12T00:19:33.557507Z","shell.execute_reply":"2021-12-12T00:19:33.564749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#torch_data.Dataset in pytorch, in order to return itertively the 8 sliced images (T1wCE image) for un patient\nclass DataRetrieverT1wCE(torch_data.Dataset):\n    def __init__(self, paths, targets):\n        self.paths = paths\n        self.targets = targets\n          \n    def __len__(self):\n        return len(self.paths)\n    \n    def __getitem__(self, index):\n        _id = self.paths[index]\n        patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{str(_id).zfill(5)}/\"\n        channels = []\n        for t in (\"T1wCE\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            channel = []\n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        y = torch.tensor(self.targets[index], dtype=torch.float)\n        return {\"X\": torch.tensor(channel).float(), \"y\": y}","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:39.138323Z","iopub.execute_input":"2021-12-12T00:19:39.138862Z","iopub.status.idle":"2021-12-12T00:19:39.148438Z","shell.execute_reply.started":"2021-12-12T00:19:39.13883Z","shell.execute_reply":"2021-12-12T00:19:39.147325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#torch_data.Dataset in pytorch, in order to return itertively the 8 sliced images (T1wCE image) for un patient\nclass DataRetrieverT2w(torch_data.Dataset):\n    def __init__(self, paths, targets):\n        self.paths = paths\n        self.targets = targets\n          \n    def __len__(self):\n        return len(self.paths)\n    \n    def __getitem__(self, index):\n        _id = self.paths[index]\n        patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{str(_id).zfill(5)}/\"\n        channels = []\n        for t in (\"T2w\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            channel = []\n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        y = torch.tensor(self.targets[index], dtype=torch.float)\n        return {\"X\": torch.tensor(channel).float(), \"y\": y}","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:44.873865Z","iopub.execute_input":"2021-12-12T00:19:44.874439Z","iopub.status.idle":"2021-12-12T00:19:44.884188Z","shell.execute_reply.started":"2021-12-12T00:19:44.874386Z","shell.execute_reply":"2021-12-12T00:19:44.883346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#torch_data.Dataset in pytorch, in order to return itertively the 8 sliced images (T1wCE image) for un patient\nclass DataRetrieverFLAIR(torch_data.Dataset):\n    def __init__(self, paths, targets):\n        self.paths = paths\n        self.targets = targets\n          \n    def __len__(self):\n        return len(self.paths)\n    \n    def __getitem__(self, index):\n        _id = self.paths[index]\n        patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{str(_id).zfill(5)}/\"\n        channels = []\n        for t in (\"FLAIR\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            channel = []\n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        y = torch.tensor(self.targets[index], dtype=torch.float)\n        return {\"X\": torch.tensor(channel).float(), \"y\": y}","metadata":{"execution":{"iopub.status.busy":"2021-12-12T00:19:49.020911Z","iopub.execute_input":"2021-12-12T00:19:49.021459Z","iopub.status.idle":"2021-12-12T00:19:49.030803Z","shell.execute_reply.started":"2021-12-12T00:19:49.021411Z","shell.execute_reply":"2021-12-12T00:19:49.029712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#torch_data.Dataset in pytorch, in order to return itertively the 8 sliced images (T1wCE image) for un patient\nclass DataRetriever(torch_data.Dataset):\n    def __init__(self, paths, targets):\n        self.paths = paths\n        self.targets = targets\n          \n    def __len__(self):\n        return len(self.paths)\n    \n    def __getitem__(self, index):\n        _id = self.paths[index]\n        patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{str(_id).zfill(5)}/\"\n        channels = []\n        for t in (\"T1w\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            channel = []\n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        for t in (\"T2w\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        for t in (\"FLAIR\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        for t in (\"T1wCE\",): # \"T2w\", \"FLAIR\", \"T1w\", \n            #print(t)\n            t_paths = sorted(\n                glob.glob(os.path.join(patient_path, t, \"*\")), \n                key=lambda x: int(x[:-4].split(\"-\")[-1]),\n            )       \n            r= len(t_paths)//2\n            channel.append(cv2.resize(load_dicom(t_paths[r]), (256, 256)) / 255)\n        y = torch.tensor(self.targets[index], dtype=torch.float)\n        return {\"X\": torch.tensor(channel).float(), \"y\": y}","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:30:00.272849Z","iopub.execute_input":"2021-12-12T01:30:00.273207Z","iopub.status.idle":"2021-12-12T01:30:00.290249Z","shell.execute_reply.started":"2021-12-12T01:30:00.273178Z","shell.execute_reply":"2021-12-12T01:30:00.289168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_retriever_0 = DataRetriever(\n    df_train_0[\"BraTS21ID\"].values, \n    df_train_0[\"MGMT_value\"].values, \n)\n\ntrain_data_retriever_1 = DataRetriever(\n    df_train_1[\"BraTS21ID\"].values, \n    df_train_1[\"MGMT_value\"].values, \n)\n\nvalid_data_retriever_0 = DataRetriever(\n    df_valid_0[\"BraTS21ID\"].values, \n    df_valid_0[\"MGMT_value\"].values,\n)\n\nvalid_data_retriever_1 = DataRetriever(\n    df_valid_1[\"BraTS21ID\"].values, \n    df_valid_1[\"MGMT_value\"].values,\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:36:21.921455Z","iopub.execute_input":"2021-12-12T01:36:21.921883Z","iopub.status.idle":"2021-12-12T01:36:21.928995Z","shell.execute_reply.started":"2021-12-12T01:36:21.921848Z","shell.execute_reply":"2021-12-12T01:36:21.927752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_data_retriever_T1w_0)","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:28:05.78971Z","iopub.execute_input":"2021-12-12T01:28:05.790345Z","iopub.status.idle":"2021-12-12T01:28:05.796688Z","shell.execute_reply.started":"2021-12-12T01:28:05.790294Z","shell.execute_reply":"2021-12-12T01:28:05.795885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_retriever_T1w_0[1]['X'].shape # 8 images of n.0 patient ","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:28:25.345388Z","iopub.execute_input":"2021-12-12T01:28:25.345796Z","iopub.status.idle":"2021-12-12T01:28:25.577669Z","shell.execute_reply.started":"2021-12-12T01:28:25.345759Z","shell.execute_reply":"2021-12-12T01:28:25.57669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super().__init__()\n        \n        # load a pre-trained EfficientNet model \n        self.net = efficientnet_pytorch.EfficientNet.from_name(\"efficientnet-b0\")\n        checkpoint = torch.load(\"../input/efficientnet-pytorch/efficientnet-b0-08094119.pth\")\n        self.net.load_state_dict(checkpoint)\n        \n        # modifier the first layer for input of grey image, (if we return rgb in DataSet, it not needed, but changer to grey by opencv2 is to slow)\n        self.net._conv_stem = nn.Conv2d(\n          1, 32, kernel_size=(3, 3), stride=(2, 2), bias=False\n        )\n        n_features = self.net._fc.in_features\n        self.net._fc = nn.Linear(in_features=n_features, out_features=100, bias=True) #modifier the last layer to have a feature vector \n    \n    def forward(self, x):\n        out = self.net(x)\n        return out","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:36:39.791201Z","iopub.execute_input":"2021-12-12T01:36:39.791879Z","iopub.status.idle":"2021-12-12T01:36:39.80027Z","shell.execute_reply.started":"2021-12-12T01:36:39.791827Z","shell.execute_reply":"2021-12-12T01:36:39.799439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model().to(device)","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:36:43.986729Z","iopub.execute_input":"2021-12-12T01:36:43.987362Z","iopub.status.idle":"2021-12-12T01:36:44.148706Z","shell.execute_reply.started":"2021-12-12T01:36:43.987311Z","shell.execute_reply":"2021-12-12T01:36:44.147739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create torch_data.DataLoader to prepare training process\n\ntrain_loader_0 = torch_data.DataLoader(\n    train_data_retriever_0,\n    batch_size=4,\n    shuffle=True,\n    num_workers=4,\n)\n\ntrain_loader_0 = torch_data.DataLoader(\n    train_data_retriever_0,\n    batch_size=4,\n    shuffle=True,\n    num_workers=4,\n)\n\ntrain_loader_1 = torch_data.DataLoader(\n    train_data_retriever_1,\n    batch_size=4,\n    shuffle=True,\n    num_workers=4,\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:37:13.429513Z","iopub.execute_input":"2021-12-12T01:37:13.429923Z","iopub.status.idle":"2021-12-12T01:37:13.435523Z","shell.execute_reply.started":"2021-12-12T01:37:13.429891Z","shell.execute_reply":"2021-12-12T01:37:13.434743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get one batch\nx = next(iter(train_loader_0))","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:37:23.514486Z","iopub.execute_input":"2021-12-12T01:37:23.515211Z","iopub.status.idle":"2021-12-12T01:37:25.333083Z","shell.execute_reply.started":"2021-12-12T01:37:23.515163Z","shell.execute_reply":"2021-12-12T01:37:25.331913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x[\"X\"].shape","metadata":{"execution":{"iopub.status.busy":"2021-12-12T01:37:27.610864Z","iopub.execute_input":"2021-12-12T01:37:27.611381Z","iopub.status.idle":"2021-12-12T01:37:27.618627Z","shell.execute_reply.started":"2021-12-12T01:37:27.611347Z","shell.execute_reply":"2021-12-12T01:37:27.617679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = []\n\nfor im in x[\"X\"]:\n    im = im.reshape(4,1,256,256).to(device) # make grey image (256, 256) to have one channel\n    feature = model(im)\n    #print(feature.shape)\n    features.append(feature)\n    \n#features = torch.stack(features)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T14:49:56.825158Z","iopub.execute_input":"2021-12-11T14:49:56.825533Z","iopub.status.idle":"2021-12-11T14:49:58.149066Z","shell.execute_reply.started":"2021-12-11T14:49:56.825501Z","shell.execute_reply":"2021-12-11T14:49:58.147928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(features)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T14:50:00.983753Z","iopub.execute_input":"2021-12-11T14:50:00.984114Z","iopub.status.idle":"2021-12-11T14:50:00.99142Z","shell.execute_reply.started":"2021-12-11T14:50:00.984082Z","shell.execute_reply":"2021-12-11T14:50:00.990039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loss to define and training process to code ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#validation, \n\n\ntrain_data_retriever = DataRetriever(\n    df_train[\"BraTS21ID\"].values, \n    df_train[\"MGMT_value\"].values, \n)\n\nvalid_data_retriever = DataRetriever(\n    df_valid[\"BraTS21ID\"].values, \n    df_valid[\"MGMT_value\"].values,\n)\n\ntrain_loader = torch_data.DataLoader(\n    train_data_retriever,\n    batch_size=8,\n    shuffle=True,\n    num_workers=8,\n)\n\nvalid_loader = torch_data.DataLoader(\n    valid_data_retriever, \n    batch_size=8,\n    shuffle=False,\n    num_workers=8,\n)\n\nmodel = Model()\nmodel.to(device)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-22T07:17:05.670877Z","iopub.execute_input":"2021-10-22T07:17:05.671183Z","iopub.status.idle":"2021-10-22T07:27:11.332021Z","shell.execute_reply.started":"2021-10-22T07:17:05.671158Z","shell.execute_reply":"2021-10-22T07:27:11.330772Z"},"trusted":true},"execution_count":null,"outputs":[]}]}