{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import Necessary libraries","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nimport pydicom\nimport numpy as np\nimport cv2\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:10.661541Z","iopub.execute_input":"2023-06-13T00:24:10.662921Z","iopub.status.idle":"2023-06-13T00:24:10.670443Z","shell.execute_reply.started":"2023-06-13T00:24:10.662837Z","shell.execute_reply":"2023-06-13T00:24:10.668941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install torchvision","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:10.673159Z","iopub.status.idle":"2023-06-13T00:24:21.820016Z","shell.execute_reply.started":"2023-06-13T00:24:10.673632Z","shell.execute_reply":"2023-06-13T00:24:21.818677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install torch torchvision torchaudio torchmetrics","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:21.821785Z","iopub.execute_input":"2023-06-13T00:24:21.822175Z","iopub.status.idle":"2023-06-13T00:24:33.17758Z","shell.execute_reply.started":"2023-06-13T00:24:21.822139Z","shell.execute_reply":"2023-06-13T00:24:33.176117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install numpy==1.22.0","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:33.181397Z","iopub.execute_input":"2023-06-13T00:24:33.182016Z","iopub.status.idle":"2023-06-13T00:24:44.638465Z","shell.execute_reply.started":"2023-06-13T00:24:33.181957Z","shell.execute_reply":"2023-06-13T00:24:44.636829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:44.645377Z","iopub.execute_input":"2023-06-13T00:24:44.645809Z","iopub.status.idle":"2023-06-13T00:24:44.653137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torchvision\nimport torchmetrics\nimport pytorch_lightning as pl\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom pytorch_lightning.loggers import TensorBoardLogger","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:44.654987Z","iopub.execute_input":"2023-06-13T00:24:44.655436Z","iopub.status.idle":"2023-06-13T00:24:44.665486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read the labels dataframe.","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-13T00:24:44.667387Z","iopub.execute_input":"2023-06-13T00:24:44.667821Z","iopub.status.idle":"2023-06-13T00:24:44.725446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:44.727264Z","iopub.execute_input":"2023-06-13T00:24:44.727665Z","iopub.status.idle":"2023-06-13T00:24:44.744062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"While looking at labels, we can see that there is a **patientid** column at first which is a unique patient id.\n\nColumns`x,y, left, right` represents the `x` and `y` coordinates and the `left` and `right` position of the classified pneumonia region.\n\nThe final column `Target` is used to denote whether patient is suffering from pneumonia or not.","metadata":{}},{"cell_type":"code","source":"# Dropping Duplicates\n\nlabels = labels.drop_duplicates(\"patientId\")","metadata":{"execution":{"iopub.status.busy":"2023-06-13T00:24:44.746335Z","iopub.execute_input":"2023-06-13T00:24:44.746677Z","iopub.status.idle":"2023-06-13T00:24:44.761422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's create a `ROOT_PATH` that takes a path to our train dataset.\n\nAdditionally, let's define a `SAVE_PATH` that defines path to the place where our processed images are saved.","metadata":{}},{"cell_type":"code","source":"ROOT_PATH = Path(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/\")\nSAVE_PATH = Path(\"Processed\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at some example images.\n\nLet's create 3x3 subplots to view 9 images along with their labels.","metadata":{}},{"cell_type":"code","source":"fig, axis = plt.subplots(3, 3, figsize = (9, 9))\nc = 0\nfor i in range(3):\n    for j in range(3):\n        # Start reading the files.\n        patient_id = labels.patientId.iloc[c]\n        \n        # Create a path to the dicom file of this particular patient/\n        dcm_path = ROOT_PATH/patient_id\n        \n        # Add .dcm extension\n        dcm_path = dcm_path.with_suffix(\".dcm\")\n        \n        # Reading the dicom file\n        dcm = pydicom.read_file(dcm_path).pixel_array\n        \n        # Extract the labels of the particular patient from labels dataframe\n        label = labels['Target'].iloc[c]\n        \n        # Visualize the image\n        axis[i][j].imshow(dcm, cmap = \"bone\")\n        axis[i][j].set_title(label)\n        \n        c += 1\n        \n         \n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###  Preprocessing Images","metadata":{}},{"cell_type":"markdown","source":"First, we standardize each image pixels by dividing it by 255.\n\nAlso, our images are way too large for current neural network architectures to process. So, we need to resize them to shape of 224x224.\n\nTo use less space when storing the images, we convert them to `float16`","metadata":{}},{"cell_type":"code","source":"# Initialization \nsums, sums_squared = 0, 0\n\n# Loop over all patient ids\n# To decide if a data is used as training data or validation data\n# We can use the enumerate \n\nfor c, patient_id, in enumerate(tqdm(labels.patientId)):\n    # Start reading the files.\n    patient_id = labels.patientId.iloc[c]\n        \n    # Create a path to the dicom file of this particular patient/\n    dcm_path = ROOT_PATH/patient_id\n\n    # Add .dcm extension\n    dcm_path = dcm_path.with_suffix(\".dcm\")\n\n    # Reading the dicom file and dividing the pixel array by 255\n    dcm = pydicom.read_file(dcm_path).pixel_array\n    \n    # Resizing the image and converting it's type\n    dcm_array = cv2.resize(dcm, (224, 224)).astype(np.float16)\n    \n    # Storing the label in 'label'\n    label = labels.Target.iloc[c]\n    \n    # Identify training/validation\n    train_or_val = \"train\" if c < 24015 else \"val\"\n    \n    # Save preprocessed images\n    current_save_path = SAVE_PATH/train_or_val/str(label)\n    current_save_path.mkdir(parents = True, exist_ok = True)\n    np.save(current_save_path/patient_id, dcm_array)\n    \n    \n    # Update sums and sums_squared\n    normalizer = 224*224\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = 0.49\nstd = 0.24","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train and Validation Dataset","metadata":{}},{"cell_type":"markdown","source":"To load the generate the data in required format, we can make use of dataset class.","metadata":{}},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, we can define our train and validation transform:","metadata":{}},{"cell_type":"code","source":"# Transforms\n\ntrain_transforms = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std),\n    transforms.RandomAffine(degrees = (-5, 5), translate = (0, 0.05), scale = (0.9, 1.1)),\n    transforms.RandomResizedCrop((224, 224), scale = (0.35, 1))\n])\n\nval_transforms = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DataLoader\n\ntrain_dataset = torchvision.datasets.DatasetFolder(\"/kaggle/working/Processed/train/\", loader = load_file, extensions = \"npy\", transform = train_transforms)\n\nval_dataset = torchvision.datasets.DatasetFolder(\"/kaggle/working/Processed/val/\", loader = load_file, extensions = \"npy\", transform = val_transforms)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axis = plt.subplots(2, 2, figsize=(9, 9))\n\nfor i in range(2):\n    for j in range(2):\n        random_index = np.random.randint(0, 24015)\n        x_ray, label = train_dataset[random_index]\n        axis[i][j].imshow(x_ray[0], cmap = \"bone\")\n        axis[i][j].set_title(label)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\nnum_workers = 4","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataLoader(train_dataset, batch_size = batch_size, num_workers = num_workers, shuffle = True)\nval_loader = DataLoader(val_dataset, batch_size = batch_size, num_workers = num_workers, shuffle = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(train_dataset.targets, return_counts = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our dataset is heavily imbalanced. So, we will be doing weighted loss.","metadata":{}},{"cell_type":"code","source":"# Observing the resnext50_32x4d architecture\n\ntorchvision.models.resnext50_32x4d()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PneumoniaModel(pl.LightningModule):\n    \n    # Constructor\n    def __init__(self):\n        super(PneumoniaModel, self).__init__()\n        self.model = torchvision.models.resnext50_32x4d()\n        self.model.conv1 = torch.nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n        self.model.fc = torch.nn.Linear(in_features = 512, out_features = 1, bias = True)\n        \n        self.optimizer = torch.optim.Adam(self.model.parameters(), lr = 1e-4)\n        self.loss_fn = torch.nn.BCEWithLogitsLoss(pos_weight = torch.tensor([3]))\n        \n        self.train_acc = torchmetrics.Accuracy(task = \"binary\")\n        self.val_acc = torchmetrics.Accuracy(task = \"binary\")\n        #self.test_acc = torchmetrics.Accuracy()\n        \n        \n        \n    # Activation\n    def forward(self, data):\n        pred = self.model(data)\n        return pred\n    \n    def training_step(self, batch, batch_idx):\n        x_ray, label = batch\n        label = label.float()\n        pred = self(x_ray)[:, 0]\n        loss = self.loss_fn(pred, label)\n        \n        self.log(\"Train Loss\", loss)\n        self.log(\"Step Train ACC\", self.train_acc(torch.sigmoid(pred), label.int()))\n        \n        return loss\n    \n    \n    \n    def on_train_epoch_end(self):\n        self.log(\"Train ACC\", self.train_acc.compute())\n        \n    \n    def validation_step(self, batch, batch_idx):\n        x_ray, label = batch\n        label = label.float()\n        pred = self(x_ray)[:, 0]\n        loss = self.loss_fn(pred, label)\n        \n        self.log(\"Val Loss\", loss)\n        self.log(\"Step Val ACC\", self.val_acc(torch.sigmoid(pred), label.int()))\n\n        \n        \n       \n    def on_validation_epoch_end(self):\n        self.log(\"Val ACC\", self.val_acc.compute())\n        \n        \n    def configure_optimizers(self):\n        return[self.optimizer]\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = PneumoniaModel()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_callback = ModelCheckpoint(\n    monitor = \"Val ACC\",\n    save_top_k = 1,\n    mode = \"max\",\n    dirpath = \"./saved_models\"\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configure the trainer\ntrainer = pl.Trainer(logger = TensorBoardLogger(save_dir = \"./logs\"), log_every_n_steps = 64, callbacks = checkpoint_callback, max_epochs = 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\ntrainer.fit(model, train_loader, val_loader)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if GPU is available\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = PneumoniaModel.load_from_checkpoint(\"/kaggle/working/saved_models/epoch=0-step=375.ckpt\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.to(device)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nlabels = []\n\nwith torch.no_grad():\n    for data, label in tqdm(val_dataset):\n        data = data.to(device).float().unsqueeze(0)\n        \n        # Calculate probabilities\n        pred = torch.sigmoid(model(data)[0].cpu())\n        \n        preds.append(pred)\n        labels.append(label)\npreds = torch.tensor(preds)\nlabels = torch.tensor(labels).int()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Accuracy\nacc = torchmetrics.Accuracy(task = \"binary\")(preds, labels)\n\n# Precision\nprecision = torchmetrics.Precision(task = \"binary\")(preds, labels)\n\n# Recall\nrecall = torchmetrics.Recall(task = \"binary\")(preds, labels)\n\n# Confusion Matrix\ncm = torchmetrics.ConfusionMatrix(num_classes=2, task = \"binary\")(preds, labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Accuracy : {acc}\")\nprint(f\"Precision : {precision}\")\nprint(f\"Recall : {recall}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Class Activation Map\nLearning deep features for discriminative localization.","metadata":{}},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_transforms = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(0.49, 0.248)\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_dataset = torchvision.datasets.DatasteFolder(\"Processes/val\", loader = load_file, extensions = 'npy', transform = val_transform)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_model = torchvision.models.resnext50_32x4d()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(temp_model.children())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(temp_model.children())[:-2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.nn.sequential(*list(temp_model.children())[:-2])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"class PneumoniaModel(pl.LightningMod\n    \n    \n        ","metadata":{}},{"cell_type":"code","source":"class PneumoniaModel(pl.LightningModule):\n    \n    def __init__(self):\n        super(PneumoniaModel, self).__init__()\n        self.model.conv1 = torch.nn.Conv2d(1, 64, kernel_size=(7, 7), stride = (2, 2), padding = (3, 3), bias = False)\n        self.model.fc = torch.nn.Linear(in_)features=512, out_features=1)\n        \n        self.feature_map = torch.nn.Sequential(*list(temp_model.children())[:-2])\n        \n        \n    def forward(self, data):\n        feature_map = self.feature_map(data)\n        avg_pool_output = torch.nn.functional.adaptive_avg_pool2d(input=feature_map, output_size = (1, 1))\n        ave_output_flattened = torch.flatten(avg_pool_output)\n        pred = self.model.fc(avg_pool_flattened)\n        return pred, feature_map","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = PneumoniaModel.load_from_checkpoint(\"weights/weig.....0\", strict = False)\nmodel.eval():","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cam(model, img):\n    with torch.no_grad():\n        pred, features = model(img.unsqueeze(0))\n    features = features.reshape((512, 49))\n    weight_params = list(model.model.fc.parameters())[0]\n    weight = weight_params[0].detach()\n    \n    cam = torch.matmul(weight, features)\n    cam_img = cam.reshape(7, 7).cpu()\n    return cam_img, torch.sigmoid(pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize(img, cam, pred):\n    img = img[0]\n    cam = transforms.functional.resize(cam.unsqueeze(0), (224, 224))[0]\n    \n    fig, axis = plt.subplots(1, 2)\n    axis[0].imshow(img, cmap = 'bone')\n    axis[1].imshow(img, cmap= 'bone')\n    axis[1].imshow(cam, alpha=0.5, cmap=\"jet\")\n    plt.title(pred>0.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = val_dataset[-6][0]\nactivation_map, pred = cam(model, img)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize(img, activation_map, pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}