{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## EfficientNet + Quantile Regression Model in Pytorch\n\nThis notebook generates predictions using Images and tabular data. \n\n### Acknowledgements \n\n* efficientnets-quantile-regression-inference: https://www.kaggle.com/leoisleo1/efficientnets-quantile-regression-inference\n\n* Training EfficientNet with pytorch: https://www.kaggle.com/noelmat/training-efficientnet-with-pytorch\n\n* melanoma-pytorch-starter-efficientnet: https://www.kaggle.com/nroman/melanoma-pytorch-starter-efficientnet","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nfrom torchvision import models\nfrom pathlib import Path\nPath.ls = lambda x: list(x.iterdir())\n\nimport cv2 \nimport pydicom\nfrom tqdm import tqdm\nfrom matplotlib import pyplot as plt\nfrom torchvision import transforms\n\nfrom torch import nn\n# from efficientnet_pytorch import EfficientNet\n# from efficientnet_pytorch.utils import MemoryEfficientSwish\nimport warnings\n\nimport random\nfrom torch.optim import Adam\nfrom torch.optim.lr_scheduler import OneCycleLR, ReduceLROnPlateau\nimport pydicom\nfrom pathlib import Path\nPath.ls = lambda x: list(x.iterdir())\nimport sys\n\nfrom sklearn.model_selection import GroupKFold\nfrom torch.utils.data import DataLoader, Subset\nfrom torch.optim.lr_scheduler import StepLR\nfrom datetime import datetime, timedelta\nfrom time import time\nimport torch.nn.functional as F\nimport copy\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-12T11:07:19.600767Z","iopub.execute_input":"2021-06-12T11:07:19.601097Z","iopub.status.idle":"2021-06-12T11:07:19.614030Z","shell.execute_reply.started":"2021-06-12T11:07:19.601061Z","shell.execute_reply":"2021-06-12T11:07:19.613147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"package_path = '../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master'\nsys.path.append(package_path)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:24.149861Z","iopub.execute_input":"2021-06-12T11:07:24.150196Z","iopub.status.idle":"2021-06-12T11:07:24.154680Z","shell.execute_reply.started":"2021-06-12T11:07:24.150163Z","shell.execute_reply":"2021-06-12T11:07:24.153529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install resnet_pytorch","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:25.175189Z","iopub.execute_input":"2021-06-12T11:07:25.175520Z","iopub.status.idle":"2021-06-12T11:07:25.179762Z","shell.execute_reply.started":"2021-06-12T11:07:25.175488Z","shell.execute_reply":"2021-06-12T11:07:25.178453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\n","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:25.455124Z","iopub.execute_input":"2021-06-12T11:07:25.456154Z","iopub.status.idle":"2021-06-12T11:07:25.505145Z","shell.execute_reply.started":"2021-06-12T11:07:25.456106Z","shell.execute_reply":"2021-06-12T11:07:25.504009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"warnings.simplefilter('ignore')\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark =True\n    \nseed_everything(42)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:26.453801Z","iopub.execute_input":"2021-06-12T11:07:26.454117Z","iopub.status.idle":"2021-06-12T11:07:26.462498Z","shell.execute_reply.started":"2021-06-12T11:07:26.454085Z","shell.execute_reply":"2021-06-12T11:07:26.461689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    def __init__(self):\n        self.FOLDS = 2\n        self.EPOCHS = 1\n        self.DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else 'cpu')\n        self.TRAIN_BS = 32\n        self.VALID_BS = 128\n        self.model_type = 'efficientnet-b3'\n        self.loss_fn = nn.L1Loss()\n        \nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:26.657990Z","iopub.execute_input":"2021-06-12T11:07:26.658317Z","iopub.status.idle":"2021-06-12T11:07:26.737411Z","shell.execute_reply.started":"2021-06-12T11:07:26.658287Z","shell.execute_reply":"2021-06-12T11:07:26.736453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('/kaggle/input/osic-pulmonary-fibrosis-progression/')\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:26.890977Z","iopub.execute_input":"2021-06-12T11:07:26.891374Z","iopub.status.idle":"2021-06-12T11:07:26.901184Z","shell.execute_reply.started":"2021-06-12T11:07:26.891341Z","shell.execute_reply":"2021-06-12T11:07:26.899951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(path/'train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:28.488140Z","iopub.execute_input":"2021-06-12T11:07:28.488471Z","iopub.status.idle":"2021-06-12T11:07:28.518921Z","shell.execute_reply.started":"2021-06-12T11:07:28.488441Z","shell.execute_reply":"2021-06-12T11:07:28.517954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(np.nonzero(np.array(train_df['Patient'] == 'ID00011637202177653955184',dtype=float))[0], axis=0).reset_index(drop=True)\ntrain_df = train_df.drop(np.nonzero(np.array(train_df['Patient'] == 'ID00052637202186188008618',dtype=float))[0], axis=0).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:29.310070Z","iopub.execute_input":"2021-06-12T11:07:29.310386Z","iopub.status.idle":"2021-06-12T11:07:29.321113Z","shell.execute_reply.started":"2021-06-12T11:07:29.310357Z","shell.execute_reply":"2021-06-12T11:07:29.320369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"def get_tab(df):\n    vector = [(df.Weeks.values[0] - 30 )/30]\n    \n    if df.Sex.values[0] == 'Male':\n       vector.append(0)\n    else:\n       vector.append(1)\n    \n    if df.SmokingStatus.values[0] == 'Never smoked':\n        vector.extend([0,0])\n    elif df.SmokingStatus.values[0] == 'Ex-smoker':\n        vector.extend([1,1])\n    elif df.SmokingStatus.values[0] == 'Currently smokes':\n        vector.extend([0,1])\n    else:\n        vector.extend([1,0])\n    return np.array(vector) ","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:31.301273Z","iopub.execute_input":"2021-06-12T11:07:31.301608Z","iopub.status.idle":"2021-06-12T11:07:31.309773Z","shell.execute_reply.started":"2021-06-12T11:07:31.301563Z","shell.execute_reply":"2021-06-12T11:07:31.308901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TAB = {}\nTARGET = {}\nPerson = []\n\nfor i, p in tqdm(enumerate(train_df.Patient.unique())):\n    sub = train_df.loc[train_df.Patient == p]\n    fvc = sub.FVC.values\n    weeks = sub.Weeks.values\n    c = np.vstack([weeks, np.ones(len(weeks))]).T\n    a, b = np.linalg.lstsq(c, fvc)[0]\n    \n    TARGET[p] = a\n    TAB[p] = get_tab(sub)\n    Person.append(p)\n\nPerson = np.array(Person)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:34.126712Z","iopub.execute_input":"2021-06-12T11:07:34.127038Z","iopub.status.idle":"2021-06-12T11:07:34.368577Z","shell.execute_reply.started":"2021-06-12T11:07:34.127007Z","shell.execute_reply":"2021-06-12T11:07:34.367718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read dicom image","metadata":{}},{"cell_type":"code","source":"def get_img(path):\n    d = pydicom.dcmread(path)\n    return cv2.resize(d.pixel_array / 2**11, (512, 512))","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-06-12T11:07:35.864059Z","iopub.execute_input":"2021-06-12T11:07:35.864421Z","iopub.status.idle":"2021-06-12T11:07:35.870199Z","shell.execute_reply.started":"2021-06-12T11:07:35.864382Z","shell.execute_reply":"2021-06-12T11:07:35.869063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset class","metadata":{}},{"cell_type":"code","source":"class Dataset:\n    def __init__(self, path, df, tabular, targets, mode , folder = 'train' ):\n        self.df = df\n        self.tabular = tabular\n        self.targets = targets\n        self.folder = folder\n        self.mode = mode\n        self.path = path\n        self.transform = transforms.Compose([\n            transforms.ToTensor()\n        ])\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self,idx):\n        row = self.df.loc[idx,:]\n        pid = row['Patient']\n        # Path to record\n        record = self.path/self.folder/pid\n        # select image id\n        try: \n            \n            img_id =  np.random.choice(len(record.ls()))\n            \n            img = get_img(record.ls()[img_id])\n            img = self.transform(img)\n            tab = torch.from_numpy(self.tabular[pid]).float()\n            if self.mode == 'train':\n                target = torch.tensor(self.targets[pid])\n                return (img,tab), target\n            else:\n                return (img,tab)\n        except Exception as e:\n            print(e)\n            print(pid, img_id)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:38.327300Z","iopub.execute_input":"2021-06-12T11:07:38.327665Z","iopub.status.idle":"2021-06-12T11:07:38.339156Z","shell.execute_reply.started":"2021-06-12T11:07:38.327626Z","shell.execute_reply":"2021-06-12T11:07:38.337987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Custom(Dataset):\n    def __init__(self, path, df, tabular, targets, mode , folder = 'train' ):\n        self.df = df\n        self.tabular = tabular\n        self.targets = targets\n        self.folder = folder\n        self.mode = mode\n        self.path = path\n        self.transform = transforms.Compose([\n            transforms.ToTensor()\n        ])\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self,idx):\n        row = self.df.loc[idx,:]\n        pid = row['Patient']\n        # Path to record\n        record = self.path/self.folder/pid/\"1.dcm\"\n        # select image id\n        try: \n            \n\n            \n            img = get_img(record)\n            img = self.transform(img)\n            tab = torch.from_numpy(self.tabular[pid]).float()\n            if self.mode == 'train':\n                target = torch.tensor(self.targets[pid])\n                return (img,tab), target\n            else:\n                return (img,tab)\n        except Exception as e:\n            print(e)\n            print(pid, img_id)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:15:35.653349Z","iopub.execute_input":"2021-06-12T11:15:35.653692Z","iopub.status.idle":"2021-06-12T11:15:35.664438Z","shell.execute_reply.started":"2021-06-12T11:15:35.653659Z","shell.execute_reply":"2021-06-12T11:15:35.663370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def collate_fn(b):\n    xs, ys = zip(*b)\n    imgs, tabs = zip(*xs)\n    return (torch.stack(imgs).float(),torch.stack(tabs).float()),torch.stack(ys).float()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:40.620367Z","iopub.execute_input":"2021-06-12T11:07:40.620735Z","iopub.status.idle":"2021-06-12T11:07:40.626357Z","shell.execute_reply.started":"2021-06-12T11:07:40.620698Z","shell.execute_reply":"2021-06-12T11:07:40.625509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Architecture","metadata":{}},{"cell_type":"code","source":"pretrained_model = {\n    'efficientnet-b0': '../input/efficientnet-pytorch/efficientnet-b0-08094119.pth',\n    'efficientnet-b3': '../input/efficientnet-pytorch/efficientnet-b3-c8376fa2.pth'\n}","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:41.908284Z","iopub.execute_input":"2021-06-12T11:07:41.908629Z","iopub.status.idle":"2021-06-12T11:07:41.914064Z","shell.execute_reply.started":"2021-06-12T11:07:41.908575Z","shell.execute_reply":"2021-06-12T11:07:41.913077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class OSIC_Model(nn.Module):\n    def __init__(self,eff_name='efficienet-b0'):\n        super().__init__()\n        self.input = nn.Conv2d(1,3,kernel_size=3,padding=1,stride=2)\n        self.bn = nn.BatchNorm2d(3)\n        #self.model = EfficientNet.from_pretrained(f'efficientnet-{eff_name}-c8376fa2.pth')\n        self.model = EfficientNet.from_name(eff_name)\n        self.model.load_state_dict(torch.load(pretrained_model[eff_name]))\n        self.model._fc = nn.Linear(1536, 500, bias=True)\n        self.meta = nn.Sequential(nn.Linear(4, 500),\n                                  nn.BatchNorm1d(500),\n                                  nn.ReLU(),\n                                  nn.Dropout(p=0.2),\n                                  nn.Linear(500,250),\n                                  nn.BatchNorm1d(250),\n                                  nn.ReLU(),\n                                  nn.Dropout(p=0.2))\n        self.output = nn.Linear(500+250, 1)\n        self.relu = nn.ReLU()\n    \n    def forward(self, x,tab):\n        x = self.relu(self.bn(self.input(x)))\n        x = self.model(x)\n        tab = self.meta(tab)\n        x = torch.cat([x, tab],dim=1)\n        return self.output(x)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:42.794840Z","iopub.execute_input":"2021-06-12T11:07:42.795171Z","iopub.status.idle":"2021-06-12T11:07:42.806276Z","shell.execute_reply.started":"2021-06-12T11:07:42.795140Z","shell.execute_reply":"2021-06-12T11:07:42.805441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Kfold splits","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\ndef get_split_idxs(n_folds=5):\n    kv = KFold(n_splits=n_folds)\n    splits = []\n    for i,(train_idx, valid_idx) in enumerate(kv.split(Person)):\n        splits.append((train_idx, valid_idx))\n        \n    return splits","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:45.269503Z","iopub.execute_input":"2021-06-12T11:07:45.269860Z","iopub.status.idle":"2021-06-12T11:07:45.275978Z","shell.execute_reply.started":"2021-06-12T11:07:45.269828Z","shell.execute_reply":"2021-06-12T11:07:45.274807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splits = get_split_idxs(n_folds=config.FOLDS)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:46.222889Z","iopub.execute_input":"2021-06-12T11:07:46.223211Z","iopub.status.idle":"2021-06-12T11:07:46.227258Z","shell.execute_reply.started":"2021-06-12T11:07:46.223178Z","shell.execute_reply":"2021-06-12T11:07:46.226438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_loop(model, dl, opt, sched, device, loss_fn):\n    model.train()\n    for X,y in dl:\n        imgs = X[0].to(device)\n        tabs = X[1].to(device)\n        y = y.to(device)\n        outputs = model(imgs, tabs)\n        loss = loss_fn(outputs.squeeze(), y)\n        opt.zero_grad()\n        loss.backward()\n        opt.step()\n        if sched is not None:\n            sched.step()\n            \n\ndef eval_loop(model, dl, device, loss_fn):\n    model.eval()\n    final_outputs = []\n    final_loss = []\n    with torch.no_grad():\n        for X,y in dl:\n            imgs = X[0].to(device)\n            tabs = X[1].to(device)\n            y=y.to(device)\n\n            outputs = model(imgs, tabs)\n            loss = loss_fn(outputs.squeeze(), y)\n\n            final_outputs.extend(outputs.detach().cpu().numpy().tolist())\n            final_loss.append(loss.detach().cpu().numpy())\n        \n    return final_outputs, final_loss","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:47.578455Z","iopub.execute_input":"2021-06-12T11:07:47.578805Z","iopub.status.idle":"2021-06-12T11:07:47.593072Z","shell.execute_reply.started":"2021-06-12T11:07:47.578774Z","shell.execute_reply":"2021-06-12T11:07:47.592109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from functools import partial\n\ndef apply_mod(m,f):\n    f(m)\n    for l in m.children(): apply_mod(l,f)\n\ndef set_grad(m,b):\n    if isinstance(m, (nn.Linear, nn.BatchNorm2d)): return \n    if hasattr(m, 'weight'):\n        for p in m.parameters(): p.requires_grad_(b)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:49.025874Z","iopub.execute_input":"2021-06-12T11:07:49.026186Z","iopub.status.idle":"2021-06-12T11:07:49.032547Z","shell.execute_reply.started":"2021-06-12T11:07:49.026155Z","shell.execute_reply":"2021-06-12T11:07:49.031314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = {}\nfor i in range(config.FOLDS):\n    models[i] = OSIC_Model(config.model_type)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:49.898447Z","iopub.execute_input":"2021-06-12T11:07:49.898803Z","iopub.status.idle":"2021-06-12T11:07:51.847165Z","shell.execute_reply.started":"2021-06-12T11:07:49.898771Z","shell.execute_reply":"2021-06-12T11:07:51.846216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k,v in models.items():\n    apply_mod(v.model, partial(set_grad, b=False))","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:52.737572Z","iopub.execute_input":"2021-06-12T11:07:52.737935Z","iopub.status.idle":"2021-06-12T11:07:52.747988Z","shell.execute_reply.started":"2021-06-12T11:07:52.737902Z","shell.execute_reply":"2021-06-12T11:07:52.747097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### View some training Images","metadata":{}},{"cell_type":"code","source":"train = train_df.loc[train_df['Patient'].isin(Person[:21])].reset_index(drop=True)\ntrain_ds = Dataset(path, train, TAB, TARGET, mode='train')\ntrain_dl = torch.utils.data.DataLoader(\n    dataset=train_ds,\n    batch_size=config.TRAIN_BS,\n    shuffle=True,\n    collate_fn=collate_fn        \n)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:07:54.161098Z","iopub.execute_input":"2021-06-12T11:07:54.161407Z","iopub.status.idle":"2021-06-12T11:07:54.170584Z","shell.execute_reply.started":"2021-06-12T11:07:54.161375Z","shell.execute_reply":"2021-06-12T11:07:54.169837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig=plt.figure(figsize=(8, 8))\ncolumns = 4\nrows = 4\ni=1\nfor X,y in train_dl:\n    pass\nj=0\nfor i in range(1, columns*rows +1):\n    img = np.array(X[0][j].permute(1,2,0))\n    img = cv2.cvtColor(img ,cv2.COLOR_GRAY2RGB)\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img)\n    j += 1\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:08:00.194310Z","iopub.execute_input":"2021-06-12T11:08:00.194670Z","iopub.status.idle":"2021-06-12T11:08:07.663161Z","shell.execute_reply.started":"2021-06-12T11:08:00.194635Z","shell.execute_reply":"2021-06-12T11:08:07.662232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"history = []","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:08:07.664580Z","iopub.execute_input":"2021-06-12T11:08:07.664924Z","iopub.status.idle":"2021-06-12T11:08:07.669340Z","shell.execute_reply.started":"2021-06-12T11:08:07.664891Z","shell.execute_reply":"2021-06-12T11:08:07.668485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (train_idx, valid_idx) in enumerate(splits):\n    print(f\"===================Fold : {i} ================\")\n\n    train = train_df.loc[train_df['Patient'].isin(Person[train_idx])].reset_index(drop=True)\n    valid = train_df.loc[train_df['Patient'].isin(Person[valid_idx])].reset_index(drop=True)\n\n\n    train_ds = Dataset(path, train, TAB, TARGET, mode= 'train')\n    train_dl = torch.utils.data.DataLoader(\n        dataset=train_ds,\n        batch_size=config.TRAIN_BS,\n        shuffle=True,\n        collate_fn=collate_fn        \n    )\n\n    valid_ds = Dataset(path, valid, TAB, TARGET, mode='train')\n    valid_dl = torch.utils.data.DataLoader(\n        dataset=valid_ds,\n        batch_size=config.VALID_BS,\n        shuffle=False,\n        collate_fn=collate_fn\n    )\n\n    model = models[i]\n    model.to(config.DEVICE)\n    lr=1e-3\n    momentum = 0.9\n    \n    num_steps = len(train_dl)\n    optimizer = Adam(model.parameters(), lr=lr,weight_decay=0.1)\n    scheduler = OneCycleLR(optimizer, \n                           max_lr=lr,\n                           epochs=config.EPOCHS,\n                           steps_per_epoch=num_steps\n                           )\n    sched = ReduceLROnPlateau(optimizer,\n                              verbose=True,\n                              factor=0.1)\n    losses = []\n    for epoch in range(config.EPOCHS):\n        print(f\"=================EPOCHS {epoch+1}================\")\n        train_loop(model, train_dl, optimizer, scheduler, config.DEVICE,config.loss_fn)\n        metrics = eval_loop(model, valid_dl,config.DEVICE,config.loss_fn)\n        total_loss = np.array(metrics[1]).mean()\n        losses.append(total_loss)\n        print(\"Loss ::\\t\", total_loss)\n        sched.step(total_loss)\n        \n    model.to('cpu')\n    history.append(losses)\n    \n    \n        ","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:08:14.813235Z","iopub.execute_input":"2021-06-12T11:08:14.813565Z","iopub.status.idle":"2021-06-12T11:09:52.795760Z","shell.execute_reply.started":"2021-06-12T11:08:14.813533Z","shell.execute_reply":"2021-06-12T11:09:52.794867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold1=history[0]\nfold2=history[1]\nplt.plot(np.linspace(0,5,5),fold1,label='Fold1')\nplt.plot(np.linspace(0,5,5),fold2,label='Fold2')\nplt.xlabel('Epochs - EfficientNet')\nplt.ylabel('Mean L1 Loss for gradient of the curve ')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T17:48:45.54052Z","iopub.execute_input":"2021-06-10T17:48:45.540901Z","iopub.status.idle":"2021-06-10T17:48:45.687953Z","shell.execute_reply.started":"2021-06-10T17:48:45.540861Z","shell.execute_reply":"2021-06-10T17:48:45.687103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k, m in models.items():\n    torch.save(m.state_dict(), f'fold_{k}.pth')","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:09:58.607142Z","iopub.execute_input":"2021-06-12T11:09:58.607498Z","iopub.status.idle":"2021-06-12T11:09:58.732566Z","shell.execute_reply.started":"2021-06-12T11:09:58.607464Z","shell.execute_reply":"2021-06-12T11:09:58.731738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction & Submission","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nsub = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:09:59.994814Z","iopub.execute_input":"2021-06-12T11:09:59.995130Z","iopub.status.idle":"2021-06-12T11:10:00.016827Z","shell.execute_reply.started":"2021-06-12T11:09:59.995101Z","shell.execute_reply":"2021-06-12T11:10:00.016078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=test_df.loc[:1,:]","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:10:00.959799Z","iopub.execute_input":"2021-06-12T11:10:00.960121Z","iopub.status.idle":"2021-06-12T11:10:00.966436Z","shell.execute_reply.started":"2021-06-12T11:10:00.960090Z","shell.execute_reply":"2021-06-12T11:10:00.963686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:10:02.214881Z","iopub.execute_input":"2021-06-12T11:10:02.215213Z","iopub.status.idle":"2021-06-12T11:10:02.227341Z","shell.execute_reply.started":"2021-06-12T11:10:02.215182Z","shell.execute_reply":"2021-06-12T11:10:02.226107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data= []\nfor i in range(1):\n    for j in range(-12, 134):\n        test_data.append([test_df['Patient'][i],j,test_df['Age'][i],test_df['Sex'][i],test_df['SmokingStatus'][i],test_df['FVC'][i],test_df['Percent'][i],str(test_df.iloc[0])+'_'+str(j)])\n\ntest_data = pd.DataFrame(test_data, columns=['Patient','Weeks','Age','Sex','SmokingStatus','FVC','Percent','Patient_Week'])","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:10:10.481189Z","iopub.execute_input":"2021-06-12T11:10:10.481531Z","iopub.status.idle":"2021-06-12T11:10:10.673031Z","shell.execute_reply.started":"2021-06-12T11:10:10.481499Z","shell.execute_reply":"2021-06-12T11:10:10.672071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head(150)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:10:15.820249Z","iopub.execute_input":"2021-06-12T11:10:15.820608Z","iopub.status.idle":"2021-06-12T11:10:15.847318Z","shell.execute_reply.started":"2021-06-12T11:10:15.820563Z","shell.execute_reply":"2021-06-12T11:10:15.846460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TAB_test = {}\n\nPerson_test = []\n\nfor i, p in tqdm(enumerate(test_data.Patient.unique())):\n    sub = test_data.loc[test_data.Patient == p]\n\n    #weeks = sub.Weeks.values\n    #c = np.vstack([weeks, np.ones(len(weeks))]).T\n\n    TAB_test[p] = get_tab(sub)\n    Person_test.append(p)\n\nPerson_test = np.array(Person_test)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:10:23.574372Z","iopub.execute_input":"2021-06-12T11:10:23.574746Z","iopub.status.idle":"2021-06-12T11:10:23.587561Z","shell.execute_reply.started":"2021-06-12T11:10:23.574711Z","shell.execute_reply":"2021-06-12T11:10:23.586475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TAB_proj = {}\n\n#Person_proj = []\n\n#for i, p in tqdm(enumerate(proj_data.Patient.unique())):\n    #sub = proj_data.loc[proj_data.Patient == p]\n\n    #weeks = sub.Weeks.values\n    #c = np.vstack([weeks, np.ones(len(weeks))]).T\n\n    #TAB_proj[p] = get_tab(sub)\n    #Person_proj.append(p)\n\n#Person_proj = np.array(Person_proj)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T17:48:46.250965Z","iopub.execute_input":"2021-06-10T17:48:46.251338Z","iopub.status.idle":"2021-06-10T17:48:46.256451Z","shell.execute_reply.started":"2021-06-10T17:48:46.251298Z","shell.execute_reply":"2021-06-10T17:48:46.255427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def collate_fn_test(b):\n    imgs, tabs = zip(*b)\n    return (torch.stack(imgs).float(),torch.stack(tabs).float())","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:10:30.192674Z","iopub.execute_input":"2021-06-12T11:10:30.193000Z","iopub.status.idle":"2021-06-12T11:10:30.198783Z","shell.execute_reply.started":"2021-06-12T11:10:30.192971Z","shell.execute_reply":"2021-06-12T11:10:30.197617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET = {}\ntest = test_data\ntest_ds = Custom(path, test_data, TAB_test,TARGET, mode= 'test')\ntest_dl = torch.utils.data.DataLoader(\n    dataset=test_ds,\n    batch_size=128,\n    shuffle=True,\n    collate_fn=collate_fn_test        \n)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:15:45.808860Z","iopub.execute_input":"2021-06-12T11:15:45.809181Z","iopub.status.idle":"2021-06-12T11:15:45.815704Z","shell.execute_reply.started":"2021-06-12T11:15:45.809152Z","shell.execute_reply":"2021-06-12T11:15:45.814788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(test_data))","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:15:47.172644Z","iopub.execute_input":"2021-06-12T11:15:47.172970Z","iopub.status.idle":"2021-06-12T11:15:47.177875Z","shell.execute_reply.started":"2021-06-12T11:15:47.172940Z","shell.execute_reply":"2021-06-12T11:15:47.176658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_predictions= np.zeros((146,1))\n\nfor i in range(len(models)):\n    \n    predictions = []\n    model = models[i]\n    model = model.to(config.DEVICE)\n    model.load_state_dict(torch.load('./fold_' +str(i)+'.pth'))\n    model.eval()\n    with torch.no_grad():\n        for X in test_dl:\n            imgs = X[0].to(config.DEVICE)\n            tabs = X[1].to(config.DEVICE)\n\n            pred = model(imgs, tabs)\n\n            predictions.extend(pred.detach().cpu().numpy().tolist())\n    avg_predictions += predictions","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:15:48.044234Z","iopub.execute_input":"2021-06-12T11:15:48.044570Z","iopub.status.idle":"2021-06-12T11:15:51.624658Z","shell.execute_reply.started":"2021-06-12T11:15:48.044538Z","shell.execute_reply":"2021-06-12T11:15:51.623244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = avg_predictions / len(models)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:15:53.818010Z","iopub.execute_input":"2021-06-12T11:15:53.818339Z","iopub.status.idle":"2021-06-12T11:15:53.822440Z","shell.execute_reply.started":"2021-06-12T11:15:53.818299Z","shell.execute_reply":"2021-06-12T11:15:53.821424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fvc = []\nconf = []\npercent=[]\nfor i in range(len(test_data)):\n    p =test_data['Patient'][i]\n    good_fvc=(test_df.loc[test_df.Patient==p]['FVC']*100/(test_df.loc[test_df.Patient==p]['Percent']))\n    B_test = predictions[i][0] * test_df.Weeks.values[test_df.Patient == p][0]\n    cur_fvc=predictions[i][0] * test_data['Weeks'][i] + test_data['FVC'][i] - B_test\n    fvc.append(predictions[i][0] * test_data['Weeks'][i] + test_data['FVC'][i] - B_test)\n    conf.append(test_data['Percent'][i] + abs(predictions[i][0]) * abs(test_df.Weeks.values[test_df.Patient == p][0] - test_data['Weeks'][i]))\n    percent.append((cur_fvc*100)/good_fvc)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:01.701498Z","iopub.execute_input":"2021-06-12T11:16:01.701863Z","iopub.status.idle":"2021-06-12T11:16:01.977035Z","shell.execute_reply.started":"2021-06-12T11:16:01.701825Z","shell.execute_reply":"2021-06-12T11:16:01.976273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsubmission = test_data[['Patient_Week']]\n","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:02.854609Z","iopub.execute_input":"2021-06-12T11:16:02.854924Z","iopub.status.idle":"2021-06-12T11:16:02.860611Z","shell.execute_reply.started":"2021-06-12T11:16:02.854893Z","shell.execute_reply":"2021-06-12T11:16:02.859709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:03.633626Z","iopub.execute_input":"2021-06-12T11:16:03.634032Z","iopub.status.idle":"2021-06-12T11:16:03.650227Z","shell.execute_reply.started":"2021-06-12T11:16:03.633993Z","shell.execute_reply":"2021-06-12T11:16:03.649376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subm ={}\nids=[]\nweek=[]\nfvc_1=[]\npercent_1=[]\nour_result=pd.DataFrame()\nfor i in range(len(submission)):\n    subm[submission['Patient_Week'][i]]=[float(fvc[i]),float(percent[i])]\n    id=submission['Patient_Week'][i].split(\"_\")[0]\n    week.append(submission['Patient_Week'][i].split(\"_\")[1])\n    ids.append(id)\n    fvc_1.append(float(fvc[i]))\n    percent_1.append(float(percent[i]))","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:05.523309Z","iopub.execute_input":"2021-06-12T11:16:05.523659Z","iopub.status.idle":"2021-06-12T11:16:05.539162Z","shell.execute_reply.started":"2021-06-12T11:16:05.523623Z","shell.execute_reply":"2021-06-12T11:16:05.538347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"our_result=pd.DataFrame()\nour_result['ID']=ids\nour_result['FVC']=fvc_1\nour_result['percent']=percent_1\nour_result['week']=week","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:06.835330Z","iopub.execute_input":"2021-06-12T11:16:06.835677Z","iopub.status.idle":"2021-06-12T11:16:06.844496Z","shell.execute_reply.started":"2021-06-12T11:16:06.835641Z","shell.execute_reply":"2021-06-12T11:16:06.843397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"our_result.to_csv('our_result.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-12T11:16:08.186262Z","iopub.execute_input":"2021-06-12T11:16:08.186616Z","iopub.status.idle":"2021-06-12T11:16:08.626170Z","shell.execute_reply.started":"2021-06-12T11:16:08.186567Z","shell.execute_reply":"2021-06-12T11:16:08.625436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['FVC'] = sub['FVC'].astype(float)\nfor i in range(len(sub)):\n    id = sub['Patient_Week'][i]\n    print(id)\n    sub['FVC'][i]= float(subm[id][0])","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:32:23.981939Z","iopub.execute_input":"2021-06-09T14:32:23.982422Z","iopub.status.idle":"2021-06-09T14:32:24.233145Z","shell.execute_reply.started":"2021-06-09T14:32:23.982387Z","shell.execute_reply":"2021-06-09T14:32:24.232224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:28:15.387081Z","iopub.execute_input":"2021-06-09T14:28:15.387479Z","iopub.status.idle":"2021-06-09T14:28:15.401355Z","shell.execute_reply.started":"2021-06-09T14:28:15.387446Z","shell.execute_reply":"2021-06-09T14:28:15.400104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('result.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T04:50:01.00113Z","iopub.execute_input":"2021-06-09T04:50:01.001574Z","iopub.status.idle":"2021-06-09T04:50:01.449077Z","shell.execute_reply.started":"2021-06-09T04:50:01.001539Z","shell.execute_reply":"2021-06-09T04:50:01.448372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Regression Model","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.066498Z","iopub.status.idle":"2021-06-01T17:00:51.067444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.069619Z","iopub.status.idle":"2021-06-01T17:00:51.070809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Inference","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.072725Z","iopub.status.idle":"2021-06-01T17:00:51.073896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## NeuralNet model","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.075765Z","iopub.status.idle":"2021-06-01T17:00:51.076884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training model","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.078701Z","iopub.status.idle":"2021-06-01T17:00:51.079846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generating submission","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.081686Z","iopub.status.idle":"2021-06-01T17:00:51.082824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.084565Z","iopub.status.idle":"2021-06-01T17:00:51.085492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ensemble (Simple Blend)","metadata":{}},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.087486Z","iopub.status.idle":"2021-06-01T17:00:51.088662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.090503Z","iopub.status.idle":"2021-06-01T17:00:51.091597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.093568Z","iopub.status.idle":"2021-06-01T17:00:51.094541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.096505Z","iopub.status.idle":"2021-06-01T17:00:51.09753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.099398Z","iopub.status.idle":"2021-06-01T17:00:51.100296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.102164Z","iopub.status.idle":"2021-06-01T17:00:51.103264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-01T17:00:51.105072Z","iopub.status.idle":"2021-06-01T17:00:51.105909Z"},"trusted":true},"execution_count":null,"outputs":[]}]}