{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-04T16:09:31.635549Z","iopub.execute_input":"2024-07-04T16:09:31.636029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset\nimport torchvision.transforms as transforms\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:07:30.31738Z","iopub.execute_input":"2024-07-04T17:07:30.318169Z","iopub.status.idle":"2024-07-04T17:07:30.323027Z","shell.execute_reply.started":"2024-07-04T17:07:30.318135Z","shell.execute_reply":"2024-07-04T17:07:30.321957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\npath2labels = r'/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\nlabels_df = pd.read_csv(path2labels)\nlabels_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:04.366253Z","iopub.execute_input":"2024-07-04T16:12:04.36687Z","iopub.status.idle":"2024-07-04T16:12:04.587535Z","shell.execute_reply.started":"2024-07-04T16:12:04.366838Z","shell.execute_reply":"2024-07-04T16:12:04.58657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:05.610297Z","iopub.execute_input":"2024-07-04T16:12:05.610953Z","iopub.status.idle":"2024-07-04T16:12:05.62034Z","shell.execute_reply.started":"2024-07-04T16:12:05.610921Z","shell.execute_reply":"2024-07-04T16:12:05.619363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport os\nfrom PIL import Image, ImageDraw","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:06.778288Z","iopub.execute_input":"2024-07-04T16:12:06.778998Z","iopub.status.idle":"2024-07-04T16:12:06.783389Z","shell.execute_reply.started":"2024-07-04T16:12:06.778966Z","shell.execute_reply":"2024-07-04T16:12:06.782381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n#get ids for malignant images\nmalignant_ids = labels_df.loc[labels_df['label']==1]['id'].values\n\npath_train = r'/kaggle/input/histopathologic-cancer-detection/train'\nplt.rcParams['figure.figsize'] = (10, 10)\nplt.subplots_adjust(wspace=0, hspace=0)\nnrows, ncols = 3,3\n# Display images\ncolor = True\nfor i, id_ in enumerate(malignant_ids[:nrows*ncols]):\n    full_filenames = os.path.join(path_train, id_ + '.tif')\n#     load image\n    img = Image.open(full_filenames)\n    # draw a 32*32 box\n    draw = ImageDraw.Draw(img)\n    draw.rectangle(((32, 32), (64, 64)),outline=\"green\")\n    plt.subplot(nrows, ncols, i+1)\n    if color is True:\n        plt.imshow(np.array(img))\n    else:\n        plt.imshow(np.array(img)[:,:,0],cmap=\"gray\")\n        plt.axis('off')\n\n    ","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-07-04T16:12:08.922252Z","iopub.execute_input":"2024-07-04T16:12:08.922865Z","iopub.status.idle":"2024-07-04T16:12:10.938647Z","shell.execute_reply.started":"2024-07-04T16:12:08.922832Z","shell.execute_reply":"2024-07-04T16:12:10.937607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.manual_seed(18)\n\nclass HistoCancer(Dataset):\n    def __init__(self, data_dir, transform, data_type='train'):\n        path_data = os.path.join(data_dir, data_type)\n        filenames = os.listdir(path_data)\n        self.fullnames = [os.path.join(path_data, i) for i in filenames]\n        # data labels\n        csv_filename = data_type + '_labels.csv'\n        path2csvlabels = os.path.join(data_dir, csv_filename)\n        labels_df = pd.read_csv(path2csvlabels)\n#         set labels index to id\n        labels_df.set_index('id', inplace=True)\n        self.transform = transform\n        self.labels = [labels_df.loc[filename[:-4]].values[0] for filename in filenames]\n        \n    def __len__(self):\n        return len(self.fullnames)\n    \n    def __getitem__(self, idx):\n        image = Image.open(self.fullnames[idx])\n        image = self.transform(image)\n        return image, self.labels[idx]","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:17.163802Z","iopub.execute_input":"2024-07-04T16:12:17.164621Z","iopub.status.idle":"2024-07-04T16:12:17.176064Z","shell.execute_reply.started":"2024-07-04T16:12:17.164582Z","shell.execute_reply":"2024-07-04T16:12:17.175083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transform = transforms.Compose([transforms.ToTensor()])\ndata_dir = r'/kaggle/input/histopathologic-cancer-detection'\nhisto_dataset = HistoCancer(data_dir, data_transform, 'train')\nprint(len(histo_dataset))","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:17.94662Z","iopub.execute_input":"2024-07-04T16:12:17.947009Z","iopub.status.idle":"2024-07-04T16:12:27.137335Z","shell.execute_reply.started":"2024-07-04T16:12:17.946981Z","shell.execute_reply":"2024-07-04T16:12:27.136402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the dataset\nlen_train = int(len(histo_dataset) * 0.8)\nlen_test = len(histo_dataset) - len_train\n\ntrain_data, test_data = torch.utils.data.random_split(histo_dataset, [len_train, len_test])","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:27.13884Z","iopub.execute_input":"2024-07-04T16:12:27.139121Z","iopub.status.idle":"2024-07-04T16:12:27.176377Z","shell.execute_reply.started":"2024-07-04T16:12:27.139097Z","shell.execute_reply":"2024-07-04T16:12:27.175698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# helper function to show images\ndef show(img,y,color=False):\n # convert tensor to numpy array\n npimg = img.numpy()\n # Convert to H*W*C shape\n npimg_tr=np.transpose(npimg, (1,2,0))\n if color==False:\n     npimg_tr=npimg_tr[:,:,0]\n     plt.imshow(npimg_tr,interpolation='nearest',cmap=\"gray\")\n else:\n # display images\n     plt.imshow(npimg_tr,interpolation='nearest')\n     plt.title(\"label: \"+str(y))","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:12:27.177357Z","iopub.execute_input":"2024-07-04T16:12:27.177805Z","iopub.status.idle":"2024-07-04T16:12:27.183846Z","shell.execute_reply.started":"2024-07-04T16:12:27.177777Z","shell.execute_reply":"2024-07-04T16:12:27.182749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create baseline models\ny_val = [y for _,y in test_data]\n\ndef accuracy(labels, out):\n    return np.sum(out==labels)/float(len(labels))\n\n# accuracy on all zero predictions\nacc_all_zeros = accuracy(y_val, np.zeros_like(y_val))\nprint(\"accuracy all zero prediction: %.2f\" %acc_all_zeros)\n\n# accuracy on all one preictions\nacc_all_ones = accuracy(y_val, np.ones_like(y_val))\nprint(\"accuracy all one prediction: %.2f\" %acc_all_ones)\n\n# accuracy random predictions\nacc_random=accuracy(y_val,np.random.randint(2,size=len(y_val)))\nprint(\"accuracy random prediction: %.2f\" %acc_random)\n\n# Helper function to findout the output size of convolution network\ndef findConv2dOutShape(H_in,W_in,conv,pool=2):\n # get conv arguments\n kernel_size=conv.kernel_size\n stride=conv.stride\n padding=conv.padding\n dilation=conv.dilation\n # Ref: https://pytorch.org/docs/stable/nn.html\n H_out=np.floor((H_in+2*padding[0]-\n dilation[0]*(kernel_size[0]-1)-1)/stride[0]+1)\n W_out=np.floor((W_in+2*padding[1]-\n dilation[1]*(kernel_size[1]-1)-1)/stride[1]+1)\n if pool:\n     H_out/=pool\n     W_out/=pool\n return int(H_out),int(W_out)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:37:17.35339Z","iopub.execute_input":"2024-07-04T16:37:17.354056Z","iopub.status.idle":"2024-07-04T16:38:59.758876Z","shell.execute_reply.started":"2024-07-04T16:37:17.354023Z","shell.execute_reply":"2024-07-04T16:38:59.757894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transforming the data\ntrain_transforms = transforms.Compose([transforms.RandomHorizontalFlip(p=0.3),\n                                      transforms.RandomVerticalFlip(p=0.3),\n                                      transforms.RandomRotation(40),\n                                      transforms.RandomResizedCrop(96,scale=(0.8,1.0),ratio=(1.0,1.0)),\n                                      transforms.ToTensor()])\n\nval_transforms = transforms.Compose([transforms.ToTensor()])\n\n# overwrite the transform functions\ntrain_data.transform=train_transforms\ntest_data.transform=val_transforms","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:13:28.86152Z","iopub.status.idle":"2024-07-04T16:13:28.861906Z","shell.execute_reply.started":"2024-07-04T16:13:28.861719Z","shell.execute_reply":"2024-07-04T16:13:28.861742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dataloaders\nfrom torch.utils.data import DataLoader\ntrain_dl = DataLoader(train_data, batch_size=64, shuffle=True)\ntest_dl = DataLoader(test_data, batch_size=64, shuffle=True)\n\nparams_model={\n \"input_shape\": (3,96,96),\n \"initial_filters\": 8,\n \"num_fc1\": 100,\n \"dropout_rate\": 0.25,\n \"num_classes\": 2,\n }","metadata":{"execution":{"iopub.status.busy":"2024-07-04T16:38:59.760461Z","iopub.execute_input":"2024-07-04T16:38:59.760787Z","iopub.status.idle":"2024-07-04T16:38:59.766304Z","shell.execute_reply.started":"2024-07-04T16:38:59.760761Z","shell.execute_reply":"2024-07-04T16:38:59.765443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating the classification model\nclass Resdown(nn.Module):\n    def __init__(self, nchnls, out_chnls=None):\n        super().__init__()\n        if not out_chnls:\n            out_chnls = nchnls\n        self.nchnls = nchnls\n        self.out = out_chnls\n        self.conv1 = nn.Conv2d(nchnls, nchnls//2, kernel_size=3, stride=1, padding=1)\n        self.conv2 = nn.Conv2d(nchnls//2, out_chnls, kernel_size=3, stride=1, padding=1)\n        self.conv3 = nn.Conv2d(nchnls, out_chnls, kernel_size=3, padding=1)\n        self.b1 = nn.BatchNorm2d(nchnls//2)\n        \n        \n    def forward(self, x):\n        x = F.relu(x)\n        x1 = self.conv3(x)\n        x_2 = F.relu(self.b1(self.conv1(x)))\n        out = self.conv2(x_2)\n        return x1+out","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:07:51.813891Z","iopub.execute_input":"2024-07-04T17:07:51.814583Z","iopub.status.idle":"2024-07-04T17:07:51.822068Z","shell.execute_reply.started":"2024-07-04T17:07:51.814555Z","shell.execute_reply":"2024-07-04T17:07:51.821236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self,params, num_layers):\n        super().__init__()\n        C_in,H_in,W_in=params[\"input_shape\"]\n        init_f=params[\"initial_filters\"]\n        num_fc1=params[\"num_fc1\"]\n        num_classes=params[\"num_classes\"]\n        self.dropout_rate=params[\"dropout_rate\"]\n        \n        self.conv1 = nn.Conv2d(C_in, init_f, kernel_size=3)\n        h, w = findConv2dOutShape(H_in, W_in, self.conv1)\n        self.b1 = nn.BatchNorm2d(init_f)\n        self.conv2 = nn.Conv2d(init_f, 2*init_f, kernel_size=3)\n        h, w = findConv2dOutShape(h, w, self.conv2)\n        self.b2 = nn.BatchNorm2d(2*init_f)\n        self.conv3 = nn.Conv2d(init_f*2, init_f*4, kernel_size=3)\n        h, w = findConv2dOutShape(h, w, self.conv3)\n        self.layers = nn.Sequential(*[Resdown(init_f*2) for i in range(num_layers)])\n        self.num_flatten = h*w*4*init_f\n        self.fc1 = nn.Linear(self.num_flatten, num_fc1)\n        self.fc2 = nn.Linear(num_fc1, num_classes)\n        \n    def forward(self, x):\n        x = F.relu(self.b1(self.conv1(x)))\n        x = F.max_pool2d(x, 2, 2)\n        x = F.relu(self.b2(self.conv2(x)))\n        x = F.max_pool2d(x, 2, 2)\n        x = F.relu(self.b2(self.layers(x)))\n        x = F.relu(self.conv3(x))\n        x = F.max_pool2d(x, 2, 2)\n        x = x.view(-1, self.num_flatten)\n        x = self.fc1(x)\n        x=F.dropout(x, self.dropout_rate, training= self.training)\n        x = self.fc2(x)\n        return F.log_softmax(x, dim=1)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:36:59.054052Z","iopub.execute_input":"2024-07-04T17:36:59.054806Z","iopub.status.idle":"2024-07-04T17:36:59.067528Z","shell.execute_reply.started":"2024-07-04T17:36:59.054775Z","shell.execute_reply":"2024-07-04T17:36:59.066609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = ('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = Net(params_model, 5)\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:37:01.911641Z","iopub.execute_input":"2024-07-04T17:37:01.912Z","iopub.status.idle":"2024-07-04T17:37:01.930328Z","shell.execute_reply.started":"2024-07-04T17:37:01.911975Z","shell.execute_reply":"2024-07-04T17:37:01.929455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install torchsummary\nfrom torchsummary import summary\nsummary(model, input_size=(3,96,96), device=device)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:37:04.72802Z","iopub.execute_input":"2024-07-04T17:37:04.728737Z","iopub.status.idle":"2024-07-04T17:37:04.813872Z","shell.execute_reply.started":"2024-07-04T17:37:04.728682Z","shell.execute_reply":"2024-07-04T17:37:04.812937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss_fn = nn.NLLLoss(reduction='sum')\nfrom torch import optim\noptimizer = optim.Adam(model.parameters(), lr=1e-3)\n\ndef get_lr(opt):\n for param_group in opt.param_groups:\n     return param_group['lr']\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n# define learning rate scheduler\nlr_scheduler = ReduceLROnPlateau(optimizer, mode='min',factor=0.5,patience=20,verbose=1)\n\n# helper function to count number of correct predictions per batch\ndef metrics_batch(output, target):\n # get output class\n pred = output.argmax(dim=1, keepdim=True)\n # compare output class with target class\n corrects=pred.eq(target.view_as(pred)).sum().item()\n return corrects\n\n# helper function to computethe loss value per batch of data \ndef loss_batch(loss_func, output, target, opt=None):\n loss = loss_func(output, target)\n with torch.no_grad():\n     metric_b = metrics_batch(output,target)\n if opt is not None:\n     opt.zero_grad()\n     loss.backward()\n     opt.step()\n return loss.item(), metric_b","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:50:12.707723Z","iopub.execute_input":"2024-07-04T17:50:12.708057Z","iopub.status.idle":"2024-07-04T17:50:12.717736Z","shell.execute_reply.started":"2024-07-04T17:50:12.708034Z","shell.execute_reply":"2024-07-04T17:50:12.716699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_train={\n \"num_epochs\": 100,\n\"optimizer\": optimizer,\n \"loss_func\": loss_fn,\n \"train_dl\": train_dl,\n \"val_dl\": test_dl,\n \"sanity_check\": True,\n \"lr_scheduler\": lr_scheduler,\n \"path2weights\": \"./models/weights.pt\"}\n\n\ndef loss_epoch(model,loss_func,dataset_dl,sanity_check=False,opt=None):\n    running_loss=0.0\n    running_metric=0.0\n    len_data=len(dataset_dl.dataset)\n    \n    for xb, yb in dataset_dl:\n        # move batch to device\n        xb=xb.to(device)\n        yb=yb.to(device)\n        # get model output\n        output=model(xb)\n        # get loss per batch\n        loss_b,metric_b=loss_batch(loss_func, output, yb, opt)\n        \n        # update running loss\n        running_loss+=loss_b\n         # update running metric\n        if metric_b is not None:\n            running_metric+=metric_b\n         # break the loop in case of sanity check\n        if sanity_check is True:\n            break\n            \n    # average loss value\n    loss=running_loss/float(len_data)\n # average metric value\n    metric=running_metric/float(len_data)\n    return loss, metric","metadata":{"execution":{"iopub.status.busy":"2024-07-04T17:59:43.64017Z","iopub.execute_input":"2024-07-04T17:59:43.640566Z","iopub.status.idle":"2024-07-04T17:59:43.648916Z","shell.execute_reply.started":"2024-07-04T17:59:43.640536Z","shell.execute_reply":"2024-07-04T17:59:43.648019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_val(model, params):\n # extract model parameters\n    num_epochs=params[\"num_epochs\"]\n    loss_func=params[\"loss_func\"]\n    opt=params[\"optimizer\"]\n    train_dl=params[\"train_dl\"]\n    val_dl=params[\"val_dl\"]\n    sanity_check=params[\"sanity_check\"]\n    lr_scheduler=params[\"lr_scheduler\"]\n    path2weights=params[\"path2weights\"]\n\n    os.makedirs(os.path.dirname(path2weights), exist_ok=True)\n    # history of loss values in each epoch\n    loss_history={\"train\": [], \"val\": []}\n     # history of metric values in each epoch\n    metric_history={\"train\": [],\"val\": []}\n    # initialize best loss to a large value\n    best_loss=float('inf')\n    # main loop\n    for epoch in range(num_epochs):\n     # get current learning rate\n        current_lr=get_lr(opt)\n        print('Epoch {}/{}, current lr={}'.format(epoch, num_epochs- 1, current_lr))\n     # train model on training dataset\n        model.train()\n        train_loss,train_metric=loss_epoch(model,loss_func,train_dl,sanity_check,opt)\n     # collect loss and metric for training dataset\n        loss_history[\"train\"].append(train_loss)\n        metric_history[\"train\"].append(train_metric)\n\n        # evaluate model on validation dataset\n        model.eval()\n        with torch.no_grad():\n            val_loss,val_metric=loss_epoch(model,loss_func,val_dl,sanity_check)\n # collect loss and metric for validation dataset\n        loss_history[\"val\"].append(val_loss)\n        metric_history[\"val\"].append(val_metric)\n\n        # store best model\n        if val_loss < best_loss:\n            best_loss = val_loss\n            best_model_wts = copy.deepcopy(model.state_dict())\n             # store weights into a local file\n            torch.save(model.state_dict(), path2weights)\n            print(\"Copied best model weights!\")\n        # learning rate schedule\n        lr_scheduler.step(val_loss)\n        if current_lr != get_lr(opt):\n            print(\"Loading best model weights!\")\n            model.load_state_dict(best_model_wts)\n\n        print(\"train loss: %.6f, dev loss: %.6f, accuracy: %.2f\"%(train_loss,val_loss,100*val_metric))\n        print(\"-\"*10)\n# load best model weights\n    model.load_state_dict(best_model_wts)\n    return model, loss_history, metric_history\n","metadata":{"execution":{"iopub.status.busy":"2024-07-04T18:17:53.034534Z","iopub.execute_input":"2024-07-04T18:17:53.035572Z","iopub.status.idle":"2024-07-04T18:17:53.046875Z","shell.execute_reply.started":"2024-07-04T18:17:53.035539Z","shell.execute_reply":"2024-07-04T18:17:53.045888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import copy\nmodel,loss_hist,metric_hist=train_val(model,params_train)","metadata":{"execution":{"iopub.status.busy":"2024-07-04T18:17:54.396921Z","iopub.execute_input":"2024-07-04T18:17:54.397595Z","iopub.status.idle":"2024-07-04T18:19:07.425558Z","shell.execute_reply.started":"2024-07-04T18:17:54.397563Z","shell.execute_reply":"2024-07-04T18:19:07.424624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}