{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30716,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-02T16:10:31.430638Z","iopub.execute_input":"2024-06-02T16:10:31.431059Z","iopub.status.idle":"2024-06-02T16:11:57.586010Z","shell.execute_reply.started":"2024-06-02T16:10:31.431026Z","shell.execute_reply":"2024-06-02T16:11:57.582196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nbase_dir = '../input/histopathologic-cancer-detection/'\nprint(os.listdir(base_dir))\n\n# Matplotlib for visualization\nimport matplotlib.pyplot as plt\nplt.style.use(\"ggplot\")\n\n# OpenCV Image Library\nimport cv2\n\n# Import PyTorch\nimport torchvision.transforms as transforms\nfrom torch.utils.data.sampler import SubsetRandomSampler\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import TensorDataset, DataLoader, Dataset\nimport torchvision\nimport torch.optim as optim\n\n# Import useful sklearn functions\nimport sklearn\nfrom sklearn.metrics import roc_auc_score, accuracy_score\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:29.302656Z","iopub.execute_input":"2024-06-02T16:19:29.303109Z","iopub.status.idle":"2024-06-02T16:19:35.007344Z","shell.execute_reply.started":"2024-06-02T16:19:29.303075Z","shell.execute_reply":"2024-06-02T16:19:35.006340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_train_df = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\nfull_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:35.009163Z","iopub.execute_input":"2024-06-02T16:19:35.009720Z","iopub.status.idle":"2024-06-02T16:19:35.352227Z","shell.execute_reply.started":"2024-06-02T16:19:35.009686Z","shell.execute_reply":"2024-06-02T16:19:35.351186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train Size: {}\".format(len(os.listdir('../input/histopathologic-cancer-detection/train/'))))\nprint(\"Test Size: {}\".format(len(os.listdir('../input/histopathologic-cancer-detection/test/'))))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:36.738655Z","iopub.execute_input":"2024-06-02T16:19:36.739266Z","iopub.status.idle":"2024-06-02T16:19:39.405830Z","shell.execute_reply.started":"2024-06-02T16:19:36.739233Z","shell.execute_reply":"2024-06-02T16:19:39.404755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_count = full_train_df.label.value_counts()\n\n%matplotlib inline\nplt.pie(labels_count, labels=['No Cancer', 'Cancer'], startangle=180, \n        autopct='%1.1f', colors=['#00ff99','#FF96A7'], shadow=True)\nplt.figure(figsize=(16,16))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:39.407444Z","iopub.execute_input":"2024-06-02T16:19:39.407821Z","iopub.status.idle":"2024-06-02T16:19:39.585307Z","shell.execute_reply.started":"2024-06-02T16:19:39.407777Z","shell.execute_reply":"2024-06-02T16:19:39.584356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(30, 6))\n# display 20 images\ntrain_imgs = os.listdir(base_dir+\"train\")\nfor idx, img in enumerate(np.random.choice(train_imgs, 20)):\n    ax = fig.add_subplot(2, 20//2, idx+1, xticks=[], yticks=[])\n    im = Image.open(base_dir+\"train/\" + img)\n    plt.imshow(im)\n    lab = full_train_df.loc[full_train_df['id'] == img.split('.')[0], 'label'].values[0]\n    ax.set_title('Label: %s'%lab)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:42.264902Z","iopub.execute_input":"2024-06-02T16:19:42.265720Z","iopub.status.idle":"2024-06-02T16:19:45.227342Z","shell.execute_reply.started":"2024-06-02T16:19:42.265690Z","shell.execute_reply":"2024-06-02T16:19:45.226244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of samples in each class\nSAMPLE_SIZE = 80000\n\n# Data paths\ntrain_path = '../input/histopathologic-cancer-detection/train/'\ntest_path = '../input/histopathologic-cancer-detection/test/'\n\n# Use 80000 positive and negative examples\ndf_negatives = full_train_df[full_train_df['label'] == 0].sample(SAMPLE_SIZE, random_state=42)\ndf_positives = full_train_df[full_train_df['label'] == 1].sample(SAMPLE_SIZE, random_state=42)\n\n# Concatenate the two dfs and shuffle them up\ntrain_df = sklearn.utils.shuffle(pd.concat([df_positives, df_negatives], axis=0).reset_index(drop=True))\n\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:50.829479Z","iopub.execute_input":"2024-06-02T16:19:50.830412Z","iopub.status.idle":"2024-06-02T16:19:50.889354Z","shell.execute_reply.started":"2024-06-02T16:19:50.830369Z","shell.execute_reply":"2024-06-02T16:19:50.888340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Our own custom class for datasets\nclass CreateDataset(Dataset):\n    def __init__(self, df_data, data_dir = './', transform=None):\n        super().__init__()\n        self.df = df_data.values\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_name,label = self.df[index]\n        img_path = os.path.join(self.data_dir, img_name+'.tif')\n        image = cv2.imread(img_path)\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:52.878677Z","iopub.execute_input":"2024-06-02T16:19:52.879349Z","iopub.status.idle":"2024-06-02T16:19:52.886256Z","shell.execute_reply.started":"2024-06-02T16:19:52.879314Z","shell.execute_reply":"2024-06-02T16:19:52.885316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transforms_train = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.RandomHorizontalFlip(p=0.4),\n    transforms.RandomVerticalFlip(p=0.4),\n    transforms.RandomRotation(20),\n    transforms.ToTensor(),\n    # We the get the following mean and std for the channels of all the images\n    #transforms.Normalize((0.70244707, 0.54624322, 0.69645334), (0.23889325, 0.28209431, 0.21625058))\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\ntrain_data = CreateDataset(df_data=train_df, data_dir=train_path, transform=transforms_train)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:54.938596Z","iopub.execute_input":"2024-06-02T16:19:54.939572Z","iopub.status.idle":"2024-06-02T16:19:54.960310Z","shell.execute_reply.started":"2024-06-02T16:19:54.939526Z","shell.execute_reply":"2024-06-02T16:19:54.959431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set Batch Size\nbatch_size = 128\n\n# Percentage of training set to use as validation\nvalid_size = 0.2\n\n# obtain training indices that will be used for validation\nnum_train = len(train_data)\nindices = list(range(num_train))\n# np.random.shuffle(indices)\nsplit = int(np.floor(valid_size * num_train))\ntrain_idx, valid_idx = indices[split:], indices[:split]\n\n# Create Samplers\ntrain_sampler = SubsetRandomSampler(train_idx)\nvalid_sampler = SubsetRandomSampler(valid_idx)\n\n# prepare data loaders (combine dataset and sampler)\ntrain_loader = DataLoader(train_data, batch_size=batch_size, sampler=train_sampler)\nvalid_loader = DataLoader(train_data, batch_size=batch_size, sampler=valid_sampler)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:56.815501Z","iopub.execute_input":"2024-06-02T16:19:56.816160Z","iopub.status.idle":"2024-06-02T16:19:56.828628Z","shell.execute_reply.started":"2024-06-02T16:19:56.816128Z","shell.execute_reply":"2024-06-02T16:19:56.827724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transforms_test = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.ToTensor(),\n    #transforms.Normalize((0.70244707, 0.54624322, 0.69645334), (0.23889325, 0.28209431, 0.21625058))\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\n# creating test data\nsample_sub = pd.read_csv(\"../input/histopathologic-cancer-detection/sample_submission.csv\")\ntest_data = CreateDataset(df_data=sample_sub, data_dir=test_path, transform=transforms_test)\n\n# prepare the test loader\ntest_loader = DataLoader(test_data, batch_size=batch_size, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:19:59.264817Z","iopub.execute_input":"2024-06-02T16:19:59.265694Z","iopub.status.idle":"2024-06-02T16:19:59.347289Z","shell.execute_reply.started":"2024-06-02T16:19:59.265660Z","shell.execute_reply":"2024-06-02T16:19:59.346557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CNN(nn.Module):\n    def __init__(self):\n        super(CNN,self).__init__()\n        # Convolutional and Pooling Layers\n        self.conv1=nn.Sequential(\n                nn.Conv2d(in_channels=3,out_channels=32,kernel_size=3,stride=1,padding=0),\n                nn.BatchNorm2d(32),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv2=nn.Sequential(\n                nn.Conv2d(in_channels=32,out_channels=64,kernel_size=2,stride=1,padding=1),\n                nn.BatchNorm2d(64),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv3=nn.Sequential(\n                nn.Conv2d(in_channels=64,out_channels=128,kernel_size=3,stride=1,padding=1),\n                nn.BatchNorm2d(128),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv4=nn.Sequential(\n                nn.Conv2d(in_channels=128,out_channels=256,kernel_size=3,stride=1,padding=1),\n                nn.BatchNorm2d(256),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv5=nn.Sequential(\n                nn.Conv2d(in_channels=256, out_channels=512, kernel_size=3, stride=1, padding=1),\n                nn.BatchNorm2d(512),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        \n        self.dropout2d = nn.Dropout2d()\n        \n        self.fc = nn.Sequential(\n            nn.Linear(512 * 3 * 3, 1024),  # Adjust the input size dynamically\n            nn.ReLU(inplace=True),\n            nn.Dropout(0.5),\n            nn.Linear(1024, 512),\n            nn.Dropout(0.5),\n            nn.Linear(512, 1),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        \"\"\"Method for Forward Prop\"\"\"\n        x = self.conv1(x)\n        x = self.conv2(x)\n        x = self.conv3(x)\n        x = self.conv4(x)\n        x = self.conv5(x)\n        x = x.view(x.size(0), -1)  # Flatten with dynamic size\n        x = self.fc(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:20:03.768839Z","iopub.execute_input":"2024-06-02T16:20:03.769698Z","iopub.status.idle":"2024-06-02T16:20:03.785241Z","shell.execute_reply.started":"2024-06-02T16:20:03.769664Z","shell.execute_reply":"2024-06-02T16:20:03.784321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check if CUDA is available\ntrain_on_gpu = torch.cuda.is_available()\n\nif not train_on_gpu:\n    print('CUDA is not available.  Training on CPU ...')\nelse:\n    print('CUDA is available!  Training on GPU ...')","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:20:06.889233Z","iopub.execute_input":"2024-06-02T16:20:06.889580Z","iopub.status.idle":"2024-06-02T16:20:06.916695Z","shell.execute_reply.started":"2024-06-02T16:20:06.889555Z","shell.execute_reply":"2024-06-02T16:20:06.915658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a complete CNN\nmodel = CNN()\nprint(model)\n\n# Move model to GPU if available\nif train_on_gpu: model.cuda()","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:20:12.246187Z","iopub.execute_input":"2024-06-02T16:20:12.246582Z","iopub.status.idle":"2024-06-02T16:20:12.514307Z","shell.execute_reply.started":"2024-06-02T16:20:12.246542Z","shell.execute_reply":"2024-06-02T16:20:12.513411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Trainable Parameters\npytorch_total_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\nprint(\"Number of trainable parameters: \\n{}\".format(pytorch_total_params))","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:20:15.434574Z","iopub.execute_input":"2024-06-02T16:20:15.435555Z","iopub.status.idle":"2024-06-02T16:20:15.441963Z","shell.execute_reply.started":"2024-06-02T16:20:15.435516Z","shell.execute_reply":"2024-06-02T16:20:15.440795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# specify loss function (categorical cross-entropy loss)\ncriterion = nn.BCELoss()\n# specify optimizer\noptimizer = torch.optim.Adam(model.parameters(), lr=0.00015)","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:20:17.767279Z","iopub.execute_input":"2024-06-02T16:20:17.767711Z","iopub.status.idle":"2024-06-02T16:20:17.780251Z","shell.execute_reply.started":"2024-06-02T16:20:17.767676Z","shell.execute_reply":"2024-06-02T16:20:17.779361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of epochs to train the model\nn_epochs = 100\n\nvalid_loss_min = np.Inf\n\n# keeping track of losses as it happen\ntrain_losses = []\nvalid_losses = []\nval_auc = []\ntest_accuracies = []\nvalid_accuracies = []\nauc_epoch = []\n\nfor epoch in range(1, n_epochs+1):\n\n    # keep track of training and validation loss\n    train_loss = 0.0\n    valid_loss = 0.0\n    \n    ###################\n    # train the model #\n    ###################\n    model.train()\n    for data, target in train_loader:\n        # move tensors to GPU if CUDA is available\n        if train_on_gpu:\n            data, target = data.cuda(), target.cuda().float()\n        target = target.view(-1, 1)\n        # clear the gradients of all optimized variables\n        optimizer.zero_grad()\n        # forward pass: compute predicted outputs by passing inputs to the model\n        output = model(data)\n        # calculate the batch loss\n        loss = criterion(output, target)\n        # backward pass: compute gradient of the loss with respect to model parameters\n        loss.backward()\n        # perform a single optimization step (parameter update)\n        optimizer.step()\n        # Update Train loss and accuracies\n        train_loss += loss.item()*data.size(0)\n        \n    ######################    \n    # validate the model #\n    ######################\n    model.eval()\n    for data, target in valid_loader:\n        # move tensors to GPU if CUDA is available\n        if train_on_gpu:\n            data, target = data.cuda(), target.cuda().float()\n        # forward pass: compute predicted outputs by passing inputs to the model\n        target = target.view(-1, 1)\n        output = model(data)\n        # calculate the batch loss\n        loss = criterion(output, target)\n        # update average validation loss \n        valid_loss += loss.item()*data.size(0)\n        #output = output.topk()\n        y_actual = target.data.cpu().numpy()\n        y_pred = output[:,-1].detach().cpu().numpy()\n        val_auc.append(roc_auc_score(y_actual, y_pred))        \n    \n    # calculate average losses\n    train_loss = train_loss/len(train_loader.sampler)\n    valid_loss = valid_loss/len(valid_loader.sampler)\n    valid_auc = np.mean(val_auc)\n    auc_epoch.append(np.mean(val_auc))\n    train_losses.append(train_loss)\n    valid_losses.append(valid_loss)\n        \n    # print training/validation statistics \n    print('Epoch: {} | Training Loss: {:.6f} | Validation Loss: {:.6f} | Validation AUC: {:.4f}'.format(\n        epoch, train_loss, valid_loss, valid_auc))\n    \n    ##################\n    # Early Stopping #\n    ##################\n    if valid_loss <= valid_loss_min:\n        print('Validation loss decreased ({:.6f} --> {:.6f}).  Saving model ...'.format(\n        valid_loss_min,\n        valid_loss))\n        torch.save(model.state_dict(), 'best_model.pt')\n        valid_loss_min = valid_loss","metadata":{"execution":{"iopub.status.busy":"2024-06-02T16:20:22.591890Z","iopub.execute_input":"2024-06-02T16:20:22.592597Z"},"trusted":true},"execution_count":null,"outputs":[]}]}