{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nfrom matplotlib.patches import Rectangle\nfrom sklearn.model_selection import KFold\nfrom torch.utils.data import Dataset\nfrom torchvision.io import read_image\nfrom torchvision.transforms import ToTensor\nfrom torchvision import transforms\nimport torch \nfrom torch.utils.data import DataLoader","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-10T03:31:44.808298Z","iopub.execute_input":"2022-09-10T03:31:44.808965Z","iopub.status.idle":"2022-09-10T03:31:46.685804Z","shell.execute_reply.started":"2022-09-10T03:31:44.808856Z","shell.execute_reply":"2022-09-10T03:31:46.684682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"orign_df = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:46.691583Z","iopub.execute_input":"2022-09-10T03:31:46.694129Z","iopub.status.idle":"2022-09-10T03:31:47.028025Z","shell.execute_reply.started":"2022-09-10T03:31:46.694080Z","shell.execute_reply":"2022-09-10T03:31:47.026950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.033204Z","iopub.execute_input":"2022-09-10T03:31:47.035627Z","iopub.status.idle":"2022-09-10T03:31:47.248529Z","shell.execute_reply.started":"2022-09-10T03:31:47.035575Z","shell.execute_reply":"2022-09-10T03:31:47.247420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.254796Z","iopub.execute_input":"2022-09-10T03:31:47.257214Z","iopub.status.idle":"2022-09-10T03:31:47.304385Z","shell.execute_reply.started":"2022-09-10T03:31:47.257172Z","shell.execute_reply":"2022-09-10T03:31:47.303347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.308681Z","iopub.execute_input":"2022-09-10T03:31:47.310838Z","iopub.status.idle":"2022-09-10T03:31:47.319510Z","shell.execute_reply.started":"2022-09-10T03:31:47.310804Z","shell.execute_reply":"2022-09-10T03:31:47.318432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.321047Z","iopub.execute_input":"2022-09-10T03:31:47.321939Z","iopub.status.idle":"2022-09-10T03:31:47.349554Z","shell.execute_reply.started":"2022-09-10T03:31:47.321903Z","shell.execute_reply":"2022-09-10T03:31:47.348592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## feature engineering","metadata":{}},{"cell_type":"code","source":"def feat_eng(df):\n    df['path'] = df['id'].apply(lambda x: '../input/hubmap-organ-segmentation/train_images/'+ str(x) + '.tiff')\n    skf = KFold(n_splits = 5, shuffle=True, random_state=72)\n    df['fold'] = -1\n    for fold, (train_idx, val_idx) in enumerate(skf.split(df)):\n        df.loc[val_idx, 'fold'] = int(fold)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.350981Z","iopub.execute_input":"2022-09-10T03:31:47.351637Z","iopub.status.idle":"2022-09-10T03:31:47.358782Z","shell.execute_reply.started":"2022-09-10T03:31:47.351601Z","shell.execute_reply":"2022-09-10T03:31:47.357895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= feat_eng(orign_df)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.360242Z","iopub.execute_input":"2022-09-10T03:31:47.361364Z","iopub.status.idle":"2022-09-10T03:31:47.378125Z","shell.execute_reply.started":"2022-09-10T03:31:47.361328Z","shell.execute_reply":"2022-09-10T03:31:47.377110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.382091Z","iopub.execute_input":"2022-09-10T03:31:47.382435Z","iopub.status.idle":"2022-09-10T03:31:47.425483Z","shell.execute_reply.started":"2022-09-10T03:31:47.382401Z","shell.execute_reply":"2022-09-10T03:31:47.424511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## encoder and decoder","metadata":{}},{"cell_type":"code","source":"def mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n\ndef rle2mask(mask_rle, shape=(3000,3000)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0::2], s[1::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.431592Z","iopub.execute_input":"2022-09-10T03:31:47.432457Z","iopub.status.idle":"2022-09-10T03:31:47.443151Z","shell.execute_reply.started":"2022-09-10T03:31:47.432420Z","shell.execute_reply":"2022-09-10T03:31:47.442149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = rle2mask(df['rle'].iloc[1,])\nb = torch.tensor(a)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T03:31:47.444592Z","iopub.execute_input":"2022-09-10T03:31:47.445183Z","iopub.status.idle":"2022-09-10T03:31:47.474979Z","shell.execute_reply.started":"2022-09-10T03:31:47.445148Z","shell.execute_reply":"2022-09-10T03:31:47.474003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization","metadata":{}},{"cell_type":"code","source":"def load_img(path):\n    img = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    img = img.astype('float32') # original is uint16\n    img = (img - img.min())/(img.max() - img.min())*255.0 # scale image to [0, 255]\n    img = img.astype('uint8')\n    return img\n\ndef shows(img, mask=None):\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8))\n    plt.imshow(img, cmap='bone')\n    \n    if mask is not None:\n        plt.imshow(mask, alpha=0.5)\n        labels = df['organ'].unique()\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:37.343832Z","iopub.execute_input":"2022-09-10T04:09:37.344408Z","iopub.status.idle":"2022-09-10T04:09:37.354173Z","shell.execute_reply.started":"2022-09-10T04:09:37.344373Z","shell.execute_reply":"2022-09-10T04:09:37.353125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_show(idx):\n    df = feat_eng(orign_df)\n    row = df.iloc[idx,]\n    ID = row['id']\n    organ = row['organ']\n    age = row['age']\n    sex = row['sex']\n    path = row['path']\n    size = (row['img_height'], row['img_width'])\n    pixel_size = row['pixel_size']\n    tissue_thickness = row['tissue_thickness']\n    rle = row['rle']\n    img = load_img(path)\n    ans = rle2mask(rle, shape = size)\n    plt.title(f'organ:{organ}\\nsex:{sex}, age:{age}\\nsize:{size}\\npixel_size:{pixel_size}μm\\ntissue_thickness:{tissue_thickness}')\n    shows(img, ans)\n    print(f'id:{ID}')","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:37.599930Z","iopub.execute_input":"2022-09-10T04:09:37.602437Z","iopub.status.idle":"2022-09-10T04:09:37.612225Z","shell.execute_reply.started":"2022-09-10T04:09:37.602398Z","shell.execute_reply":"2022-09-10T04:09:37.611315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_show(1)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:37.839690Z","iopub.execute_input":"2022-09-10T04:09:37.840232Z","iopub.status.idle":"2022-09-10T04:09:39.966296Z","shell.execute_reply.started":"2022-09-10T04:09:37.840202Z","shell.execute_reply":"2022-09-10T04:09:39.965202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset","metadata":{}},{"cell_type":"code","source":"\n_transform =transforms.Compose([transforms.ToTensor(),\n                                transforms.Resize((224,224))\n                               ]\n                                )\n                    \n\nclass HPADataset(Dataset):\n    def __init__(self,df,transform = _transform):\n        self.df = feat_eng(df)\n        self.transform = _transform\n        self.img_path = self.df['path']\n        self.rle = self.df['rle']\n        self.height = self.df['img_height']\n        self.width = self.df['img_width']\n        \n    def __len__(self):\n        return len(self.df) \n    \n    def __getitem__(self, idx):\n        self.img = cv2.imread(self.img_path.iloc[idx,])\n        self.en_mask = self.rle.iloc[idx,]\n        self.shape = (self.height.iloc[idx,], self.width.iloc[idx,])\n        self.de_mask = rle2mask(mask_rle = self.en_mask, shape = self.shape)\n        if self.transform:\n            self.image = self.transform(self.img)\n            self.label = self.transform(self.de_mask)\n        return self.image, self.label","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:39.968448Z","iopub.execute_input":"2022-09-10T04:09:39.969348Z","iopub.status.idle":"2022-09-10T04:09:39.980405Z","shell.execute_reply.started":"2022-09-10T04:09:39.969312Z","shell.execute_reply":"2022-09-10T04:09:39.979402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = HPADataset(orign_df)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:39.982059Z","iopub.execute_input":"2022-09-10T04:09:39.982924Z","iopub.status.idle":"2022-09-10T04:09:39.999886Z","shell.execute_reply.started":"2022-09-10T04:09:39.982887Z","shell.execute_reply":"2022-09-10T04:09:39.998812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image, label = a.__getitem__(1)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:40.002430Z","iopub.execute_input":"2022-09-10T04:09:40.003487Z","iopub.status.idle":"2022-09-10T04:09:40.382769Z","shell.execute_reply.started":"2022-09-10T04:09:40.003443Z","shell.execute_reply":"2022-09-10T04:09:40.381808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_loaders(fold):\n    train_df = df.query(\"fold!=@fold\").reset_index(drop=True)\n    valid_df = df.query(\"fold==@fold\").reset_index(drop=True)\n    train_dataset = HPADataset(train_df, \n#                             transforms=data_transforms['train']\n                           )\n    valid_dataset = HPADataset(valid_df,\n#                             transforms=data_transforms['valid']\n                           )\n\n    train_loader = DataLoader(train_dataset, batch_size=8)\n    valid_loader = DataLoader(valid_dataset, batch_size=8)\n    \n    return {'train': train_loader, 'val': valid_loader}\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:09:40.384199Z","iopub.execute_input":"2022-09-10T04:09:40.384578Z","iopub.status.idle":"2022-09-10T04:09:40.392239Z","shell.execute_reply.started":"2022-09-10T04:09:40.384538Z","shell.execute_reply":"2022-09-10T04:09:40.391031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prepare_loaders(2)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:57.047896Z","iopub.execute_input":"2022-09-10T04:27:57.048936Z","iopub.status.idle":"2022-09-10T04:27:57.071305Z","shell.execute_reply.started":"2022-09-10T04:27:57.048889Z","shell.execute_reply":"2022-09-10T04:27:57.070083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ----------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"!pip install segmentation_models_pytorch","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:26:54.460349Z","iopub.execute_input":"2022-09-10T04:26:54.461300Z","iopub.status.idle":"2022-09-10T04:27:03.862823Z","shell.execute_reply.started":"2022-09-10T04:26:54.461263Z","shell.execute_reply":"2022-09-10T04:27:03.861631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import segmentation_models_pytorch as smp\n\n# parameter\nENCODER = \"efficientnet-b4\"\nENCODER_WEIGHTS = \"imagenet\"\nCLASS_NUM = 1\nACTIVATION = \"softmax2d\"\n\ndevice = torch.device('cuda:0') if torch.cuda.is_available() else torch.device('cpu')\n\n# mdoel definition\nmodel = smp.UnetPlusPlus(\n    encoder_name = ENCODER, \n    encoder_weights = ENCODER_WEIGHTS, \n    classes = CLASS_NUM,\n    activation = ACTIVATION,\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:03.866661Z","iopub.execute_input":"2022-09-10T04:27:03.867002Z","iopub.status.idle":"2022-09-10T04:27:04.316045Z","shell.execute_reply.started":"2022-09-10T04:27:03.866971Z","shell.execute_reply":"2022-09-10T04:27:04.315083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.to(device)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:04.317666Z","iopub.execute_input":"2022-09-10T04:27:04.318076Z","iopub.status.idle":"2022-09-10T04:27:04.372332Z","shell.execute_reply.started":"2022-09-10T04:27:04.318039Z","shell.execute_reply":"2022-09-10T04:27:04.371352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model():\n    model = smp.UnetPlusPlus(\n    encoder_name = ENCODER, \n    encoder_weights = ENCODER_WEIGHTS, \n    classes = CLASS_NUM,\n    activation = ACTIVATION)\n    model = model.to(device)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:36:34.310052Z","iopub.execute_input":"2022-09-10T04:36:34.310639Z","iopub.status.idle":"2022-09-10T04:36:34.316217Z","shell.execute_reply.started":"2022-09-10T04:36:34.310604Z","shell.execute_reply":"2022-09-10T04:36:34.315251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import segmentation_models_pytorch as smp\n\nJaccardLoss = smp.losses.JaccardLoss(mode='multilabel')\nDiceLoss    = smp.losses.DiceLoss(mode='multilabel')\nBCELoss     = smp.losses.SoftBCEWithLogitsLoss()\nLovaszLoss  = smp.losses.LovaszLoss(mode='multilabel', per_image=False)\nTverskyLoss = smp.losses.TverskyLoss(mode='multilabel', log_loss=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:04.375088Z","iopub.execute_input":"2022-09-10T04:27:04.375458Z","iopub.status.idle":"2022-09-10T04:27:04.381485Z","shell.execute_reply.started":"2022-09-10T04:27:04.375422Z","shell.execute_reply":"2022-09-10T04:27:04.380225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(), lr = 0.8)\nscheduler = torch.optim.lr_scheduler.ExponentialLR(optimizer, gamma=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:04.384954Z","iopub.execute_input":"2022-09-10T04:27:04.385228Z","iopub.status.idle":"2022-09-10T04:27:04.395560Z","shell.execute_reply.started":"2022-09-10T04:27:04.385204Z","shell.execute_reply":"2022-09-10T04:27:04.394406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.size(3)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:04.396834Z","iopub.execute_input":"2022-09-10T04:27:04.397393Z","iopub.status.idle":"2022-09-10T04:27:04.409403Z","shell.execute_reply.started":"2022-09-10T04:27:04.397359Z","shell.execute_reply":"2022-09-10T04:27:04.408413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport copy\ndef train_one_epoch(model, dataloaders, criterion, optimizer, scheduler, num_epochs=5):\n    since = time.time()\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_acc = 0.0\n\n    for epoch in range(num_epochs):\n        print('  Epoch {}/{}'.format(epoch + 1, num_epochs))\n        print('-' * 10)\n\n        \n        for phase in ['train', 'val']:\n            if phase == 'train':\n                model.train()  # train mode\n            else:\n                model.eval()   # eval mode\n\n            running_loss = 0.0\n            running_corrects = 0\n            dataset_sizes = 0\n            datasets_bits = 0\n            \n            # iteration\n            for inputs, labels in dataloaders[phase]:\n                inputs = inputs.to(device)\n                labels = labels.to(device)\n\n                # zero grad\n                optimizer.zero_grad()\n                \n                with torch.set_grad_enabled(phase == 'train'):\n                    outputs = model(inputs)\n                    _, preds = torch.max(outputs, 1)\n                    loss = criterion(outputs, labels)\n\n                    if phase == 'train':\n                        loss.backward()\n                        optimizer.step()\n\n                # caluculation  loss\n                running_loss += loss.item() * inputs.size(0)\n                dataset_sizes += inputs.size(0)\n                datasets_bits += inputs.size(0) * inputs.size(1) * inputs.size(2) * inputs.size(3)\n                running_corrects += torch.sum(preds == labels.data)\n            if phase == 'train':\n                scheduler.step()\n            \n            \n            epoch_loss = running_loss / dataset_sizes\n            epoch_acc = running_corrects.double() / datasets_bits\n\n            print('{} Loss: {:.4f} Acc: {:.4f}'.format(\n                phase, epoch_loss, epoch_acc))\n\n            # copy model\n            if phase == 'val' and epoch_acc > best_acc:\n                best_acc = epoch_acc\n                best_model_wts = copy.deepcopy(model.state_dict())\n\n        print()\n\n    time_elapsed = time.time() - since\n    print('Training complete in {:.0f}m {:.0f}s'.format(\n        time_elapsed // 60, time_elapsed % 60))\n    print('Best val Acc: {:4f}'.format(best_acc))\n\n    # load the best weight of model\n    model.load_state_dict(best_model_wts)\n    return model, ","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:27:04.410894Z","iopub.execute_input":"2022-09-10T04:27:04.412527Z","iopub.status.idle":"2022-09-10T04:27:04.426546Z","shell.execute_reply.started":"2022-09-10T04:27:04.412490Z","shell.execute_reply":"2022-09-10T04:27:04.424995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(fold):\n    model_list = []\n    acc_list = []\n    for i in range(fold):\n        print(f'fold: {i+1}/ {fold}')\n        print('-' * 10)\n        model = make_model()\n        dataloaders = prepare_loaders(i)\n        one_model = train_one_epoch(model, dataloaders, DiceLoss, optimizer, scheduler, num_epochs=3)\n        model_list.append(one_model)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:37:03.597582Z","iopub.execute_input":"2022-09-10T04:37:03.597956Z","iopub.status.idle":"2022-09-10T04:37:03.604258Z","shell.execute_reply.started":"2022-09-10T04:37:03.597924Z","shell.execute_reply":"2022-09-10T04:37:03.603193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train(5)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T04:37:05.490988Z","iopub.execute_input":"2022-09-10T04:37:05.491656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}