{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Load Dataset from Kaggle","metadata":{"id":"NjTI-IjNNEnK"}},{"cell_type":"markdown","source":"## Install and import EfficientNet PyTorch","metadata":{"id":"3jfDmuk7lCWw"}},{"cell_type":"code","source":"!pip install efficientnet_pytorch\nfrom efficientnet_pytorch import EfficientNet","metadata":{"id":"Xyc5Gyj3lBoO","outputId":"207229c5-3afb-47fc-b251-4d36a3fbc76a","execution":{"iopub.status.busy":"2022-12-08T22:53:35.404225Z","iopub.execute_input":"2022-12-08T22:53:35.405271Z","iopub.status.idle":"2022-12-08T22:53:44.377978Z","shell.execute_reply.started":"2022-12-08T22:53:35.405212Z","shell.execute_reply":"2022-12-08T22:53:44.376788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Install and import Albumentations","metadata":{"id":"nacBIjcuTwg8"}},{"cell_type":"code","source":"from albumentations import (\n  HorizontalFlip, VerticalFlip, IAAPerspective, ShiftScaleRotate, CLAHE, RandomRotate90,\n    Transpose, ShiftScaleRotate, Blur, OpticalDistortion, GridDistortion, HueSaturationValue,\n    IAAAdditiveGaussianNoise, GaussNoise, MotionBlur, MedianBlur, RandomBrightnessContrast, IAAPiecewiseAffine,\n    IAASharpen, IAAEmboss, Flip, OneOf, Compose, ElasticTransform, RandomBrightness, RandomContrast\n)\nimport cv2\nimport numpy as np\nfrom urllib.request import urlopen\nfrom matplotlib import pyplot as plt","metadata":{"id":"l4rL3ZTBT0LE","execution":{"iopub.status.busy":"2022-12-08T22:53:44.381314Z","iopub.execute_input":"2022-12-08T22:53:44.381687Z","iopub.status.idle":"2022-12-08T22:53:44.389686Z","shell.execute_reply.started":"2022-12-08T22:53:44.381651Z","shell.execute_reply":"2022-12-08T22:53:44.388568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load packeges","metadata":{"id":"x2gD-rPDOA48"}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torch.utils.data as D\nimport torch.nn.functional as F\n\nimport torchvision\nfrom torchvision import transforms\n\nimport tqdm\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nimport math\n\n# torch.backends.cudnn.enabled = False\n\n","metadata":{"id":"N249OmROMi2c","execution":{"iopub.status.busy":"2022-12-08T22:53:44.390896Z","iopub.execute_input":"2022-12-08T22:53:44.391175Z","iopub.status.idle":"2022-12-08T22:53:44.402269Z","shell.execute_reply.started":"2022-12-08T22:53:44.391150Z","shell.execute_reply":"2022-12-08T22:53:44.401205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_data = '../input/recursion-cellular-image-classification'\ndevice = 'cuda'\nnum_classes = 1108\nbatch_size = 16\nnum_epochs = 3","metadata":{"id":"_DRj1xQiOEwy","execution":{"iopub.status.busy":"2022-12-08T22:53:44.405381Z","iopub.execute_input":"2022-12-08T22:53:44.405913Z","iopub.status.idle":"2022-12-08T22:53:44.415647Z","shell.execute_reply.started":"2022-12-08T22:53:44.405875Z","shell.execute_reply":"2022-12-08T22:53:44.414636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset / Batch Loader","metadata":{"id":"RnTTBmv0OGls"}},{"cell_type":"code","source":"class ImageDS(D.Dataset):\n    \"\"\"\n    Images have 6 channels\n    \"\"\"\n    def __init__(self, df, img_dir, mode='train', site=1, \n                 channels=[1,2,3,4,5,6],\n                 transform=None):\n        self.records = df.to_records(index=False) # convert DF to NP record array\n        self.channels = channels\n        self.site = site # get path to image\n        self.mode = mode\n        self.img_dir = img_dir\n        self.len = df.shape[0]\n        self.transform = transform\n        \n    def _load_img_as_tensor(self, file_name):\n        \"\"\"\n        convert PIL image into torch tensor\n        \"\"\" \n        with Image.open(file_name) as img:\n            return torchvision.transforms.ToTensor()(img)\n        \n    def _get_img_path(self, index, channel):\n        \"\"\"\n        from NP record records get path details of image: epxeriemnt. well, plate for the given channel \n        path example ../input/train/HUVEC-08/Plate1/I11_s1_w6.png\n        \"\"\"\n        experiment = self.records[index].experiment \n        well = self.records[index].well \n        plate = self.records[index].plate\n        site_mine = 3-self.site\n        # retirn path\n        p='/'.join([self.img_dir, self.mode, experiment, \n                         f'Plate{plate}', f'{well}_s{self.site}_w{channel}.png'])\n        import os\n        if(os.path.exists(p)):\n            return(p)\n        \n        return '/'.join([self.img_dir, self.mode, experiment, f'Plate{plate}', f'{well}_s{site_mine}_w{channel}.png'])\n    def _augment(self, img):\n        \"\"\"\n        Apply transforms from albu package \n        \"\"\"\n        #return img\n        aug = Compose([\n                RandomRotate90(p=0.5),\n                OneOf([\n                    HorizontalFlip(p=0.5),\n                    VerticalFlip(p=0.5)\n                ]),\n                GridDistortion(p=0.3),\n                OneOf([\n                    RandomBrightness(p=0.5),\n                    RandomContrast(p=0.5)\n                ])\n              ])\n\n        img = img.numpy() # convert to numpy\n        img = aug(image=img)['image']\n        img = torch.tensor(img).float() # return back to torch tensor\n        if img.shape[0] != len(self.channels):\n          img = torch.transpose(img, 0, 1)\n        return img\n\n    \n    def __getitem__(self, index):\n        # get image path for the given channel\n        paths = [self._get_img_path(index, ch) for ch in self.channels]\n        \n        # convert img into tensor and stack tensors  in the given dimension\n        img = torch.cat([self._load_img_as_tensor(img_path) for img_path in paths])\n        \n        # Apply transforms using Albu\n        if self.transform is not None:\n          img = self._augment(img)\n        \n        #print('final', img.shape)\n        if self.mode == 'train':\n            return img, self.records[index].sirna, self.records[index].experiment# return img tensor/ label/ experiment\n        else:\n            return img, self.records[index].id_code, self.records[index].experiment # ???\n        \n        \n    def __len__(self):\n        \"\"\"\n        total number of samples in the dataset\n        \"\"\"\n        return self.len\n    ","metadata":{"id":"hbbqmcYGOJWK","execution":{"iopub.status.busy":"2022-12-08T22:53:44.419260Z","iopub.execute_input":"2022-12-08T22:53:44.419712Z","iopub.status.idle":"2022-12-08T22:53:44.435545Z","shell.execute_reply.started":"2022-12-08T22:53:44.419685Z","shell.execute_reply":"2022-12-08T22:53:44.434533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%% Generate dataframes\ndf = pd.read_csv(path_data+'/train.csv')\ndf['sirna']=df['sirna'].str[6:].astype(int)\n\nfrom sklearn.preprocessing import LabelEncoder\n\n\nohe = LabelEncoder()\nohe.fit(df[['sirna']])\ndf['sirna'] = ohe.transform(df[['sirna']])\n\n#df=df[np.logical_or(df['experiment'].str[:-3]==\"HEPG2\" , df['experiment'].str[:-3]==\"HUVEC\")]\ndf_train, df_val = train_test_split(df, test_size=0.15, random_state=42, shuffle=True)\ndf_test = pd.read_csv(path_data+'/test.csv')\nprint(df_test)\n\n# ok\n\n# split df's into experiments\ndf_train_exp_dict = {}\ndf_val_exp_dict = {}\ndf_test_exp_dict = {}\ndf_train['exp'] = df_train['experiment'].str[:-3]\ndf_val['exp'] = df_val['experiment'].str[:-3]\ndf_test['exp'] = df_test['experiment'].str[:-3]\nunique_train_exp = df_train['exp'].unique()\nunique_val_exp = df_val['exp'].unique()\nunique_test_exp = df_test['exp'].unique()\n\nfor exp in unique_train_exp:\n    df_train_exp_dict[exp] = df_train[df_train['exp'] == exp]\nfor exp in unique_val_exp:\n    df_val_exp_dict[exp] = df_val[df_val['exp'] == exp]\nfor exp in unique_test_exp:\n    df_test_exp_dict[exp] = df_test[df_test['exp'] == exp]","metadata":{"id":"BIQLaZeepLew","outputId":"e4d1ecd2-01d0-4000-caf5-6c1dd392acdf","execution":{"iopub.status.busy":"2022-12-08T22:53:44.437020Z","iopub.execute_input":"2022-12-08T22:53:44.437384Z","iopub.status.idle":"2022-12-08T22:53:44.593865Z","shell.execute_reply.started":"2022-12-08T22:53:44.437351Z","shell.execute_reply":"2022-12-08T22:53:44.592035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train\nds_dict = {}\nfor exp in df_train_exp_dict:\n    ds_dict[exp] = ImageDS(df_train_exp_dict[exp], path_data, transform=True)\n    for i in range(3):\n        #Create transformed datasets\n        ds_temp = ImageDS(df_train_exp_dict[exp], path_data, transform=True)\n        ds_dict[exp] = torch.utils.data.ConcatDataset([ds_dict[exp], ds_temp])\n\n#val\nds_val_dict = {}\nfor exp in df_val_exp_dict:\n    ds_val_dict[exp] = ImageDS(df_val_exp_dict[exp], path_data, transform=None)\n\n#test\nds_test_dict = {}\nfor exp in df_test_exp_dict:\n    ds_test_dict[exp] = ImageDS(df_test_exp_dict[exp], path_data, mode='test', transform=None)\n\n\n# print(len(ds_dict['RPE']), len(ds_temp), len(ds_val_dict['RPE']), len(ds_test_dict['RPE']))\n# print(ds_dict['RPE'])\n# print(ds_dict['RPE'][0][0].shape, ds_dict['RPE'][0][0].mean(), ds_dict['RPE'][0][0].std())\n# 0 - image, 1 - label, 2 - experiment\n# ok\n","metadata":{"id":"0kSGpymLOM9J","execution":{"iopub.status.busy":"2022-12-08T22:54:01.221487Z","iopub.execute_input":"2022-12-08T22:54:01.221858Z","iopub.status.idle":"2022-12-08T22:54:01.320959Z","shell.execute_reply.started":"2022-12-08T22:54:01.221826Z","shell.execute_reply":"2022-12-08T22:54:01.319993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Batch loader\nloader_dict = {}\nfor exp in df_train_exp_dict:\n    loader_dict[exp] = D.DataLoader(ds_dict[exp], batch_size, shuffle=True, num_workers=2)\n\nval_loader_dict = {}\nfor exp in df_val_exp_dict:\n    val_loader_dict[exp] = D.DataLoader(ds_val_dict[exp], batch_size, shuffle=False, num_workers=2)\n\ntloader_dict = {}\nfor exp in df_test_exp_dict:\n    tloader_dict[exp] = D.DataLoader(ds_test_dict[exp], batch_size, shuffle=False, num_workers=2)","metadata":{"id":"R5UbJSvHON_B","execution":{"iopub.status.busy":"2022-12-08T22:54:03.704649Z","iopub.execute_input":"2022-12-08T22:54:03.705009Z","iopub.status.idle":"2022-12-08T22:54:03.712052Z","shell.execute_reply.started":"2022-12-08T22:54:03.704978Z","shell.execute_reply":"2022-12-08T22:54:03.710932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define Model","metadata":{"id":"w-8GVlpMOQF7"}},{"cell_type":"code","source":"from collections import OrderedDict\n\nclass EffNetb0(nn.Module):\n    def __init__(self, model_load_path=None):\n        super(EffNetb0, self).__init__() # deriving from Module\n        \n        model = EfficientNet.from_pretrained('efficientnet-b0', num_classes=1108)\n        for param in model.parameters():\n              param.requires_grad = True\n        \n        # change count input channels of our model\n        trained_kernel = model._conv_stem.weight\n        new_conv = nn.Sequential(nn.Conv2d(6, 32, kernel_size=(3,3), stride=(2,2), bias=False),\n                                 nn.ZeroPad2d(padding=(0,1,0,1))\n                                )\n        with torch.no_grad():\n          new_conv[0].weight[:,:] = torch.stack([torch.mean(trained_kernel,1)]*6, dim=1)\n        model._conv_stem = new_conv\n        \n        if model_load_path:\n            # current model dict\n            #model_dict = model.state_dict()\n            \n            # load pretrained state dict\n            print('Load pretrained model')\n            state_dict = torch.load(model_load_path)\n            new_state_dict = OrderedDict([(k[6:], v) if k[:1] != '_' else (k, v) for k, v in state_dict.items()])\n            model.load_state_dict(new_state_dict)\n        \n        model = model.to(device)\n        self.model = model\n        \n    def forward(self, x):\n        logits = self.model(x)\n        return logits\n","metadata":{"id":"TfQEIm0iSxy5","execution":{"iopub.status.busy":"2022-12-08T22:54:06.434832Z","iopub.execute_input":"2022-12-08T22:54:06.435204Z","iopub.status.idle":"2022-12-08T22:54:06.445539Z","shell.execute_reply.started":"2022-12-08T22:54:06.435173Z","shell.execute_reply":"2022-12-08T22:54:06.444305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#%% Define model / Load pretrained model\n# define model paths\nmodel_names = {'HEPG2': 'effnetb0_v11_010919_HEPG2.pth',\n               'HUVEC': 'effnetb0_v11_010919_HUVEC.pth',\n               'RPE': 'effnetb0_v11_010919_RPE.pth',\n              'U2OS': 'effnetb0_v11_010919_U2OS.pth'}\n\nmodels_dict = {}\nfor exp in unique_train_exp:\n    models_dict[exp] = EffNetb0()#model_names[exp])\n    \n# check the model is on device\nprint(device)\nnext(models_dict['HUVEC'].parameters()).is_cuda","metadata":{"id":"rjhnVvCb_Hjc","outputId":"17e5917a-e252-4085-b818-5c6e990fa5b4","execution":{"iopub.status.busy":"2022-12-08T22:54:09.323121Z","iopub.execute_input":"2022-12-08T22:54:09.323548Z","iopub.status.idle":"2022-12-08T22:54:14.541610Z","shell.execute_reply.started":"2022-12-08T22:54:09.323505Z","shell.execute_reply":"2022-12-08T22:54:14.540624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train model","metadata":{"id":"x_xq_clB2NlZ"}},{"cell_type":"code","source":"def evaluate_single_epoch(model, dataloader, criterion, \n                          batch_size, epoch, device=None, exp=None):\n  model.eval()\n  \n  running_loss = 0\n  running_corrects = 0\n  \n  with torch.no_grad():\n    total_size = len(dataloader.dataset)\n    total_step = math.ceil(total_size / batch_size)\n    \n    for i, data in enumerate(dataloader):\n      try:\n        if device:\n          x, y = data[0].to(device), data[1].to(device)\n        else:\n          x, y = data[0], data[1]\n\n        output = model(x)\n        target = torch.zeros_like(output, device=device)\n        target[np.arange(x.size(0)), y] = 1\n        \n        loss = criterion(output, y)\n        running_loss += loss.item()\n        # Calculate accuracy\n        preds = torch.argmax(output, 1)\n        running_corrects += torch.sum(preds == y).data.cpu().detach().numpy()\n      \n      except RuntimeError:\n        print('validation batch_size error')\n\n    # show val_loss, val_acc\n    print('---------------------------------------------------------------')\n    print('Exp: {} - Epoch: {} - Iter: {}/ {} - Val Loss: {:.5f}, Val Acc: {:.4f}%'.format(exp, epoch+1, i, total_step,\n                                                                           running_loss/total_size,\n                                                                           running_corrects/total_size))\n    \n    del loss, output, y, x, target\n      \ndef train_single_epoch(model, dataloaders, criterion, \n                       optimizer, batch_size, epoch, \n                       scheduler, model_save_path, \n                       device=None, exp=None):\n  scheduler.step()\n  model.train()\n  \n  dataloader, val_dataloader = dataloaders[0], dataloaders[1]\n  \n  running_loss = 0\n  running_corrects = 0\n  total_size = len(dataloader.dataset)\n  total_step = math.ceil(total_size / batch_size)\n\n  for i, data in enumerate(dataloader):\n    model.train()\n    # try:\n      # train\n    if device:      \n        x, y = data[0].to(device), data[1].to(device)\n    else:\n        x, y = data[0], data[1]\n    optimizer.zero_grad()\n    output = model(x)\n    \n    target = torch.zeros_like(output, device=device)\n    target[np.arange(x.size(0)), y] = 1\n    \n    loss = criterion(output, y)\n    loss.backward()\n    optimizer.step()\n\n    running_loss += loss.item()\n    # Calculate accuracy\n    preds = torch.argmax(output, 1)\n    running_corrects += torch.sum(preds == y).data.cpu().detach().numpy()    \n\n    # show train_loss/ train_acc each 50 batches\n    if i % 50 == 49:\n      print('Exp: {} - Epoch: {} - Iter: {}/ {} - Train Loss: {:.5f}, Train Acc: {:.4f}'.format(exp, \n                                                                                                epoch+1, i, \n                                                                                                total_step,\n                                                                                                running_loss/50,\n                                                                                                running_corrects/batch_size/50))\n      running_loss = 0\n      running_corrects = 0\n      \n    if i % 400 == 399:\n      # val phase\n      evaluate_single_epoch(model, val_dataloader, criterion, \n                            batch_size, epoch, device, exp)\n      # save the model\n      torch.save(model.state_dict(),model_save_path)\n      model.train()\n      \n    del loss, output, y, x, target\n        `\n    # except RuntimeError:\n    #   print('train batch_size error')\n  \ndef train_model(model, dataloaders, num_epochs, \n                criterion, optimizer, scheduler, \n                model_save_path, batch_size=1, device=None, exp=None):\n  start_epoch = 0  \n  for epoch in range(start_epoch, num_epochs):\n    # train phase\n    train_single_epoch(model, dataloaders, criterion, \n                       optimizer, batch_size, epoch, \n                       scheduler, model_save_path, \n                       device, exp)    \n    # save the model\n    torch.save(model.state_dict(),model_save_path)\n        ","metadata":{"id":"ecentmBPGj79","execution":{"iopub.status.busy":"2022-12-08T22:54:14.543650Z","iopub.execute_input":"2022-12-08T22:54:14.544339Z","iopub.status.idle":"2022-12-08T22:54:14.562680Z","shell.execute_reply.started":"2022-12-08T22:54:14.544300Z","shell.execute_reply":"2022-12-08T22:54:14.561284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nfrom torch.optim import lr_scheduler\n\ncriterion = nn.CrossEntropyLoss()\ndataloaders_dict = {}\noptimizers_dict = {}\nexpl_lr_scheduler_dict = {}\nfor exp in unique_train_exp:\n    dataloaders_dict[exp] = [loader_dict[exp], val_loader_dict[exp]]\n    optimizers_dict[exp] = torch.optim.Adam(models_dict[exp].parameters(), lr=0.00015)\n    expl_lr_scheduler_dict[exp] = lr_scheduler.StepLR(optimizers_dict[exp], step_size=1, gamma=0.5)\n\nmodel_save_path = {unique_train_exp[i]: 'effnetb0_v11_050919_{}.pth'.format(unique_train_exp[i]) \n                    for i in range(len(unique_train_exp))}","metadata":{"id":"39NLbR-1AjzT","execution":{"iopub.status.busy":"2022-12-08T22:54:22.067664Z","iopub.execute_input":"2022-12-08T22:54:22.068054Z","iopub.status.idle":"2022-12-08T22:54:22.082029Z","shell.execute_reply.started":"2022-12-08T22:54:22.068021Z","shell.execute_reply":"2022-12-08T22:54:22.081060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CUDA_LAUNCH_BLOCKING=1\n# train and validate model\nexps_to_improve = ['HUVEC', 'HEPG2', 'RPE', 'U2OS']\nfor exp in exps_to_improve:\n    train_model(models_dict[exp], \n                dataloaders_dict[exp], \n                num_epochs, \n                criterion, \n                optimizers_dict[exp], \n                expl_lr_scheduler_dict[exp], \n                model_save_path[exp],\n                batch_size=batch_size,\n                device=device,\n                exp=exp)","metadata":{"id":"g3f1-fgQ_c3F","execution":{"iopub.status.busy":"2022-12-08T22:54:24.244115Z","iopub.execute_input":"2022-12-08T22:54:24.245093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction/ Submit\n\n### Define method + check experiments","metadata":{"id":"1TDnjzuzOf40"}},{"cell_type":"code","source":"# Prediction for test\ndef prediction(model, loader, batch_size, exp):\n    preds = np.empty(0)\n    total_size = len(loader.dataset)\n    total_step = math.ceil(total_size / batch_size)\n    for i, data in enumerate(loader):\n        x = data[0].to(device)\n        output = model(x)\n        idx = output.max(dim=-1)[1].cpu().numpy()\n        preds = np.append(preds, idx, axis=0)\n        print('Exp: {} - Iter: {}/ {} - x.shape: {}'.format(exp, i, total_step, x.shape))\n          \n        del output, x\n          \n      \n        #print('test batch_szie error for shape: {}'.format(x.shape))\n    return preds","metadata":{"id":"Zuwebx2YD0-S","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_preds_dict[exp] = prediction(model, val_loader, batch_size)","metadata":{"id":"pioi27WTD47a","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define vectors of exps/ sirnas for ds_val\nexps = [x[2][:-3] for x in ds_val]\nsirnas = [x[1] for x in ds_val]","metadata":{"id":"5IKXrdWSJLhN","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exps = [x[2][:-3] for x in ds_val]","metadata":{"id":"YfXMn35vOJaw","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Map val_preds with experiments from ds_val in DF\nprint(val_preds)\n#exps1 = [x[:-3] for x in exps]\ndf_exp = pd.DataFrame({'experiment': exps, 'sirna': sirnas, 'sirna_pred': val_preds})\ndf_exp['equal'] = np.where(df_exp['sirna']==df_exp['sirna_pred'], 1, 0)\n#df_exp['experiement'] = df_exp['experiment'].str[:-3]\nprint(df_exp.shape)\nprint(df_exp.head())\n\ndf_eq = df_exp.groupby('experiment').count()\ndf_eq['count'] = df_exp.groupby('experiment').sum()['equal']\n#df_eq['equal_ratio'] = df_eq['equal'] / df['count']\n#df_eq['equal_ratio'] = df_eq.apply(lambda row: int(row.count) / int(row.equal), axis=1)\n\nprint(df_eq)","metadata":{"id":"njv0Oj4PGySU","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction/ Submit","metadata":{"id":"V3HXKhe0H5dT"}},{"cell_type":"code","source":"#%% Define model / Load pretrained model\n# define model paths\nmodel_names = {'HEPG2': 'effnetb0_v11_050919_HEPG2.pth',\n               'HUVEC': 'effnetb0_v11_050919_HUVEC.pth',\n               'RPE': 'effnetb0_v11_050919_RPE.pth',\n               'U2OS': 'effnetb0_v11_050919_U2OS.pth'}\n\nmodels_dict = {}\nfor exp in unique_train_exp:\n    models_dict[exp] = EffNetb0(model_names[exp])\n    \n# check the model is on device\nprint(device)\nnext(models_dict['RPE'].parameters()).is_cuda","metadata":{"id":"mlKJKrVTePdF","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_dict = {}\nfor exp in unique_train_exp:\n    preds_dict[exp] = prediction(models_dict[exp], tloader_dict[exp], batch_size, exp)\n    print(preds_dict[exp])\n    preds_dict[exp]=le.inverse_transform(preds_dict[exp])\n    print(preds_dict[exp])\n\n","metadata":{"id":"itgkjE8TOh_d","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate all the experiments in preds array\ncount  = 0\npreds_array = np.array([])\nfor exp in preds_dict:\n    print(len(preds_dict[exp]), type(preds_dict[exp]))\n    count += len(preds_dict[exp])\nprint(count)\n\nexps = ['HEPG2', 'HUVEC', 'RPE', 'U2OS']\npreds_array = np.concatenate((preds_dict[exps[0]], \n                              preds_dict[exps[1]],\n                              preds_dict[exps[2]],\n                              preds_dict[exps[3]]))\npreds_array.shape","metadata":{"id":"0QZRaSh-SfeW","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(path_data + '/test.csv')\nsubmission['sirna'] = preds_array.astype(int)\nsubmission.to_csv('submission.csv', index=False, columns=['id_code','sirna'])","metadata":{"id":"x1JIzTRuOjfl","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(submission)","metadata":{"id":"vIfwkmQeUhZc","trusted":true},"execution_count":null,"outputs":[]}]}