{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-21T23:36:00.949398Z","iopub.execute_input":"2022-11-21T23:36:00.949874Z","iopub.status.idle":"2022-11-21T23:36:10.926656Z","shell.execute_reply.started":"2022-11-21T23:36:00.949838Z","shell.execute_reply":"2022-11-21T23:36:10.925011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torchmetrics timm\nimport gc\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport glob\nimport seaborn as sns  \nfrom torch.utils.data import Dataset,DataLoader\nfrom torchvision import datasets, models, transforms\nimport torch\nfrom matplotlib import pyplot as plt\nimport os\nfrom cv2 import imread\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nimport torchmetrics \nimport timm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom torch import nn\nfrom torch.optim import AdamW,Adam # optmizers\nimport time\nfrom tqdm import tqdm\n\n%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:10.928839Z","iopub.execute_input":"2022-11-21T23:36:10.929345Z","iopub.status.idle":"2022-11-21T23:36:30.804461Z","shell.execute_reply.started":"2022-11-21T23:36:10.929311Z","shell.execute_reply":"2022-11-21T23:36:30.802891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = \"../input/224-224-cervical-cancer-screening/kaggle/train/train\"\nimages  =  [glob.glob(os.path.join(data_path, d, \"*.*\")) for d in os.listdir(data_path)]\ntrain_paths = np.hstack(images)\n# Additional data\nextra_1 = \"../input/224-224-cervical-cancer-screening/kaggle/additional_Type_1_v2\"\nextra_2 = \"../input/224-224-cervical-cancer-screening/kaggle/additional_Type_2_v2\"\nextra_3 = \"../input/224-224-cervical-cancer-screening/kaggle/additional_Type_3_v2\"\nimages1  =  [glob.glob(os.path.join(extra_1, d, \"*.*\")) for d in os.listdir(extra_1)]\nimages2  =  [glob.glob(os.path.join(extra_2, d, \"*.*\")) for d in os.listdir(extra_2)]\nimages3  =  [glob.glob(os.path.join(extra_3, d, \"*.*\")) for d in os.listdir(extra_3)]\ntrain_paths = np.append(train_paths, np.hstack(images1))\ntrain_paths = np.append(train_paths, np.hstack(images2))\ntrain_paths = np.append(train_paths, np.hstack(images3))\n\nprint(f'In this train set we have got a total of {len(train_paths)}')\nN_EPOCHS = 5\nOUTPUT_PATH = './'\nBATCH_SIZE = 10\n# detect and define device \ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)\ndevice = torch.device(device)\ncpu = torch.device('cpu')","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:30.806159Z","iopub.execute_input":"2022-11-21T23:36:30.806820Z","iopub.status.idle":"2022-11-21T23:36:30.859333Z","shell.execute_reply.started":"2022-11-21T23:36:30.806782Z","shell.execute_reply":"2022-11-21T23:36:30.857973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(8, 8), dpi=80)\ncolumns = 3\nrows = 1\nimg_type1 = plt.imread('../input/224-224-cervical-cancer-screening/kaggle/train/train/Type_1/0.jpg')\nfig.add_subplot(rows, columns, 1)\nplt.title(\"Type1\")\nplt.axis('off')\nplt.imshow(img_type1)\n\nimg_type2 = plt.imread('../input/224-224-cervical-cancer-screening/kaggle/train/train/Type_2/1.jpg')\nfig.add_subplot(rows, columns, 2)\nplt.title(\"Type2\")\nplt.axis('off')\nplt.imshow(img_type2)\n\nimg_type3 = plt.imread('../input/224-224-cervical-cancer-screening/kaggle/train/train/Type_3/1000.jpg')\nfig.add_subplot(rows, columns, 3)\nplt.title(\"Type3\")\nplt.axis('off')\nplt.imshow(img_type3)","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:30.862350Z","iopub.execute_input":"2022-11-21T23:36:30.863723Z","iopub.status.idle":"2022-11-21T23:36:31.228098Z","shell.execute_reply.started":"2022-11-21T23:36:30.863674Z","shell.execute_reply":"2022-11-21T23:36:31.227075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyDataset(Dataset):\n    \n    def __init__(self, paths, transform=None, train=True, size=224):\n        self.paths = paths\n        self.transform = transform\n        self.train = train\n        self.size = size\n    def __len__(self):\n        return len(self.paths)\n    \n    def __getitem__(self, idx):\n        p = self.paths[idx]\n        label = p.split(\"/\")[-2].split(\"_\")[-1]\n        image = cv2.imread(p)\n        image = cv2.cvtColor(image,cv2.COLOR_BGR2RGB)\n        return image, int(label) - 1 # label count starts from zero not 1 \n# here we are using dataframe\nclass CancerDataset(Dataset):\n    def __init__(self, df, x_col = \"image\", y_col = \"target\", augmentations = None):\n        self.df = df\n        self.features = df[x_col] # scale (greyscale) only features. do not scale target\n        self.targets = df[y_col]\n        self.augmentations = augmentations\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        image = self.features[idx].float()\n        label = self.targets[idx]\n        #swap [224,224,3]---TO---> [3,224,224]\n        image = image.permute(2, 0,1)\n        if self.augmentations is not None:\n            image = np.array(image)\n            augmented = self.augmentations(image=image)  \n\n            image = torch.from_numpy(augmented['image'])\n            image = image.float()\n            return image, label\n\n        return image, label\n\ndef norm(img):\n    img-=img.min()\n    img=img/img.max()\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:31.229243Z","iopub.execute_input":"2022-11-21T23:36:31.230370Z","iopub.status.idle":"2022-11-21T23:36:31.243337Z","shell.execute_reply.started":"2022-11-21T23:36:31.230330Z","shell.execute_reply":"2022-11-21T23:36:31.241801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = A.Compose([\n    A.OneOf([A.Blur(blur_limit=3), A.MotionBlur(blur_limit=3)]),\n    A.GaussNoise(), A.Sharpen(),\n    A.RandomBrightnessContrast(brightness_limit=0.1, contrast_limit=0.1),\n    A.Rotate(limit=10, border_mode=cv2.BORDER_CONSTANT),])    \n\ndataset = MyDataset(train_paths, transform, train=True, size=224)\ntrain_loader = DataLoader(dataset, batch_size=BATCH_SIZE)\ndf =  pd.DataFrame(columns=[\"image\",\"target\"])\niter_loader = iter(train_loader)\nfor x,y in iter_loader:\n    for i,(row,t) in enumerate(zip(x,y)):\n        df = df.append({\"image\":row, \"target\": t.item()}, ignore_index = True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:31.245529Z","iopub.execute_input":"2022-11-21T23:36:31.246044Z","iopub.status.idle":"2022-11-21T23:36:47.730337Z","shell.execute_reply.started":"2022-11-21T23:36:31.246007Z","shell.execute_reply":"2022-11-21T23:36:47.727369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Note- 0-type1, 1-type2, 2-type3\n#plotting the graph  \nsns.countplot(x='target', data=df, palette='pastel')  \nplt.show()  ","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.732538Z","iopub.status.idle":"2022-11-21T23:36:47.733606Z","shell.execute_reply.started":"2022-11-21T23:36:47.733257Z","shell.execute_reply":"2022-11-21T23:36:47.733290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## random split\ntrain_df, valid_df = train_test_split(df, test_size=0.3, random_state= 42)\ntrain_df.reset_index(inplace = True)\nvalid_df.reset_index(inplace = True)\n\n# create dataset for validation & train\ntrain_dataset = CancerDataset(train_df, augmentations = transform) \nvalid_dataset = CancerDataset(valid_df)\n\n# create dataloaders\ntrain_dataloader = DataLoader(train_dataset,\n                              batch_size = BATCH_SIZE,\n                              shuffle = False)\n\nvalid_dataloader = DataLoader(valid_dataset,\n                              batch_size = BATCH_SIZE,\n                              shuffle = False)","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.735438Z","iopub.status.idle":"2022-11-21T23:36:47.736416Z","shell.execute_reply.started":"2022-11-21T23:36:47.736111Z","shell.execute_reply":"2022-11-21T23:36:47.736140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FocalLoss(nn.Module):\n    def __init__(self, alpha=0.25, gamma=2.0):\n        super(FocalLoss, self).__init__()\n        self.gamma = gamma\n        self.alpha = alpha\n        self.cross_entropy_loss = nn.CrossEntropyLoss()\n\n    def forward(self, inputs, targets):\n        bce = self.cross_entropy_loss(inputs, targets)\n        pt = torch.exp(-bce)\n        loss = bce * self.alpha * (torch.pow((1 - pt), self.gamma))\n        return loss\n\n# in general alpha should be decreasing and gamma should be increasing.\nclass Model(nn.Module):\n    def __init__(self, model_name, pretrained = True, num_classes = 3):\n        super().__init__()\n        self.model_name = model_name\n        self.cnn = timm.create_model(self.model_name, pretrained = pretrained, num_classes = num_classes)\n\n    def forward(self, x):\n        x = self.cnn(x)\n        return x\n    \n    def train_mode(self):\n        self.best_loss = np.inf\n        self.best_epoch = 0\n        self.best_acc = 0\n        self.train_loss_history = []\n        self.train_acc_history = []\n        \n    def valid_mode(self):\n        self.valid_loss_history = []\n        self.valid_acc_history = []\n        \ndef train_one_epoch(train_loader, model, criterion, optimizer, device):\n    # switch to train mode\n    model.train()   \n    size = len(train_loader.dataset)\n    num_batches = len(train_loader)\n    loss, correct = 0, 0\n    ################################# train #################################\n    for batch, (x, y) in enumerate(train_loader):\n        start = time.time()\n        device = torch.device(device)\n        x, y = x.to(device), y.to(device)  \n        # compute predictions and loss\n        optimizer.zero_grad()\n        pred = model(x)\n        loss = criterion(pred, y.long().squeeze()) \n        current = batch * len(x)\n        # Backpropagation: only in train function, not done in validation function\n        loss.backward()\n        optimizer.step()\n        # sum correct predictions\n        y_pred, y_true = torch.argmax(pred, axis=1), y.long().squeeze()\n        correct += (y_pred == y_true).type(torch.float).sum().item()\n        end = time.time()\n        time_delta = np.round(end - start, 3)\n        # log\n        loss, current = np.round(loss.item(), 5), batch * len(x)\n    # metrics: calculate accuracy and loss for epoch (all batches)\n    correct /= size # epoch accuracy\n    loss /= num_batches # epoch loss\n    print(f\"Train: Accuracy: {(100*correct):>0.2f}%, Avg loss: {loss:>5f} \\n\")\n    model.train_loss_history.append(loss)\n    model.train_acc_history.append(correct)\n    return loss, correct\n    \ndef valid_one_epoch(valid_loader, model, criterion, device):\n    model.eval()\n    size = len(valid_loader.dataset)\n    num_batches = len(valid_loader)\n    loss, correct = 0, 0\n    ################################# validation #################################\n    with torch.no_grad(): # disable gradients\n        for batch, (x, y) in enumerate(valid_loader):\n            start = time.time()\n            device = torch.device(device)\n            x, y = x.to(device), y.to(device)\n            # compute predictions and loss\n            pred = model(x)\n            loss = criterion(pred, y.long().squeeze()) \n            current = batch * len(x)\n            # sum correct predictions\n            y_pred, y_true = torch.argmax(pred, axis=1), y.long().squeeze()\n            correct += (y_pred == y_true).type(torch.float).sum().item()\n            end = time.time()\n            time_delta = np.round(end - start, 3)\n            # log\n            loss, current = np.round(loss.item(), 5), batch * len(x)\n    # metrics: calculate accuracy and loss for epoch (all batches)\n    correct /= size # epoch accuracy\n    loss /= num_batches # epoch loss\n    model.valid_loss_history.append(loss)\n    model.valid_acc_history.append(correct)\n    print(f\"Valid: Accuracy: {(100*correct):>0.2f}%, Avg loss: {loss:>5f} \\n\")\n    return loss, correct\n\ndef train_valid(train_loader,valid_loader, model, device):\n    # Create optimizer & loss\n    model.optimizer = Adam(model.parameters(),lr=1e-4)\n    loss_fn = FocalLoss()\n    \n    print('\\n ******************************* Using backbone: ', model.model_name, \" ******************************* \\n\")\n    print('Starting Training...\\n')\n    start_train_time = time.time()\n    model.train_mode()\n    model.valid_mode()\n    for epoch in tqdm(range(0, N_EPOCHS)):\n        print(f\"\\n-------------------------------   Epoch {epoch + 1}   -------------------------------\\n\")\n        start_epoch_time = time.time()\n        # train\n        train_one_epoch(train_loader, model, loss_fn, model.optimizer, device)\n        # validation\n        valid_loss, valid_acc = valid_one_epoch(valid_loader, model, loss_fn, device)\n        # save validation loss if it was improved (reduced) & validation accuracy if it was improved (increased)\n        if valid_loss < model.best_loss and valid_acc > model.best_acc:\n            model.best_epoch = epoch + 1\n            model.best_loss = valid_loss\n            model.best_acc = valid_acc\n            # save the model's weights and biases   \n            torch.save(model.state_dict(), OUTPUT_PATH + f\"{model.model_name}_ep{model.best_epoch}.pth\")        \n            torch.save(model.state_dict(), OUTPUT_PATH + f\"{model.model_name}_ep{model.best_epoch}.pth\")\n\n        end_epoch_time = time.time()\n        time_delta = np.round(end_epoch_time - start_epoch_time, 3)\n        print(\"\\n\\nEpoch Elapsed Time: {} s\".format(time_delta))\n\n    end_train_time = time.time()\n    print(\"\\n\\nTotal Elapsed Time: {} min\".format(np.round((end_train_time - start_train_time)/60, 3)))\n    print(\"Done!\")\n\ndef plot_results(model):\n    fig = plt.figure(figsize = (18, 8))\n    fig.suptitle(f\"{model.model_name} Training Results\", fontsize = 18)\n\n    space = np.arange(1, N_EPOCHS + 1, 1)\n    if N_EPOCHS <= 20:\n        x_ticks = np.arange(1, N_EPOCHS + 1, 1)\n    else:\n        x_ticks = np.arange(1, N_EPOCHS + 1, int(N_EPOCHS/20) + 1)\n\n    # Loss plot\n    ax1 = plt.subplot(1, 2, 1) \n    ax1.plot(space, model.train_loss_history, label='Training', color = 'black')\n    ax1.plot(space, model.valid_loss_history, label='Validation', color = 'blue')\n    plt.xticks(x_ticks)\n    plt.axhline(0, linestyle = 'dashed', color = 'grey')\n    plt.axvline(model.best_epoch, linestyle = 'dashed', color = 'blue', label = 'Best val loss: ep ' + str(model.best_epoch))\n    plt.title(\"Loss\")\n    ax1.legend(frameon=False);\n\n    # Accuracy plot\n    ax2 = plt.subplot(1, 2, 2)\n    ax2.plot(space, model.train_acc_history, label='Training', color = 'black')\n    ax2.plot(space, model.valid_acc_history, label='Validation', color = 'blue')\n    plt.xticks(x_ticks)\n    plt.axhline(0.99, linestyle = 'dashed', color = 'grey')\n    plt.axvline(model.best_epoch, linestyle = 'dashed', color = 'green', label = 'Best val acc: ep ' + str(model.best_epoch))\n    plt.title(\"Accuracy\")\n    ax2.legend(frameon=False);","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.738425Z","iopub.status.idle":"2022-11-21T23:36:47.739379Z","shell.execute_reply.started":"2022-11-21T23:36:47.739080Z","shell.execute_reply":"2022-11-21T23:36:47.739109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#MobileNetV3","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.741131Z","iopub.status.idle":"2022-11-21T23:36:47.742330Z","shell.execute_reply.started":"2022-11-21T23:36:47.741977Z","shell.execute_reply":"2022-11-21T23:36:47.742045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_mobileNet = Model('mobilenetv3_large_100', pretrained = True, num_classes = 3)\nmodel_mobileNet = model_mobileNet.to(device) # move the model to GPU before constructing optimizers for it\ntrain_valid(train_dataloader,valid_dataloader, model_mobileNet, device)\nplot_results(model_mobileNet)\n\nmodel_mobileNet = model_mobileNet.to(cpu)\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.744242Z","iopub.status.idle":"2022-11-21T23:36:47.744958Z","shell.execute_reply.started":"2022-11-21T23:36:47.744706Z","shell.execute_reply":"2022-11-21T23:36:47.744732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ResNet50\n","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.746338Z","iopub.status.idle":"2022-11-21T23:36:47.746813Z","shell.execute_reply.started":"2022-11-21T23:36:47.746605Z","shell.execute_reply":"2022-11-21T23:36:47.746630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_resNet50 = Model('resnet50', pretrained = True, num_classes = 3)\nmodel_resNet50 = model_resNet50.to(device) # move the model to GPU before constructing optimizers for it\ntrain_valid(train_dataloader,valid_dataloader, model_resNet50, device)\nplot_results(model_resNet50)\n\nmodel_resNet50 = model_resNet50.to(cpu)\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.747829Z","iopub.status.idle":"2022-11-21T23:36:47.748239Z","shell.execute_reply.started":"2022-11-21T23:36:47.748028Z","shell.execute_reply":"2022-11-21T23:36:47.748046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#EfficentNet-B3","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.750968Z","iopub.status.idle":"2022-11-21T23:36:47.752247Z","shell.execute_reply.started":"2022-11-21T23:36:47.752006Z","shell.execute_reply":"2022-11-21T23:36:47.752031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"model_efficientnet_b3 = Model('efficientnet_b3_pruned', pretrained = True, num_classes = 3)\nmodel_efficientnet_b3 = model_efficientnet_b3.to(device) # move the model to GPU before constructing optimizers for it\ntrain_valid(train_dataloader,valid_dataloader, model_efficientnet_b3, device)\nplot_results(model_efficientnet_b3)\n\nmodel_efficientnet_b3 = model_efficientnet_b3.to(cpu)\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-11-21T23:36:47.753787Z","iopub.status.idle":"2022-11-21T23:36:47.754244Z","shell.execute_reply.started":"2022-11-21T23:36:47.754040Z","shell.execute_reply":"2022-11-21T23:36:47.754066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}