{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-07T15:06:47.076237Z","iopub.execute_input":"2023-06-07T15:06:47.076623Z","iopub.status.idle":"2023-06-07T15:06:53.592303Z","shell.execute_reply.started":"2023-06-07T15:06:47.076592Z","shell.execute_reply":"2023-06-07T15:06:53.591479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install torchvision ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install torch torchvision torchaudio torchmetrics","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nimport pydicom\nimport numpy as np\nimport cv2\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:06:41.594417Z","iopub.execute_input":"2023-06-07T15:06:41.594743Z","iopub.status.idle":"2023-06-07T15:06:41.949074Z","shell.execute_reply.started":"2023-06-07T15:06:41.594709Z","shell.execute_reply":"2023-06-07T15:06:41.948112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms\nimport torch\nimport torchvision\nimport torchmetrics\nimport pytorch_lightning as pl\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom pytorch_lightning.loggers import TensorBoardLogger","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:18.649594Z","iopub.execute_input":"2023-06-07T15:07:18.649952Z","iopub.status.idle":"2023-06-07T15:07:33.168012Z","shell.execute_reply.started":"2023-06-07T15:07:18.649922Z","shell.execute_reply":"2023-06-07T15:07:33.167014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:33.169852Z","iopub.execute_input":"2023-06-07T15:07:33.170201Z","iopub.status.idle":"2023-06-07T15:07:33.252501Z","shell.execute_reply.started":"2023-06-07T15:07:33.17015Z","shell.execute_reply":"2023-06-07T15:07:33.251546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head(6)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:43.380032Z","iopub.execute_input":"2023-06-07T15:07:43.380406Z","iopub.status.idle":"2023-06-07T15:07:43.40325Z","shell.execute_reply.started":"2023-06-07T15:07:43.380377Z","shell.execute_reply":"2023-06-07T15:07:43.40241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop Duplicates\n\nlabels = labels.drop_duplicates(\"patientId\")","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:44.491678Z","iopub.execute_input":"2023-06-07T15:07:44.492072Z","iopub.status.idle":"2023-06-07T15:07:44.508137Z","shell.execute_reply.started":"2023-06-07T15:07:44.492041Z","shell.execute_reply":"2023-06-07T15:07:44.506983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_PATH = Path(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/\")\nSAVE_PATH = Path(\"Processed_images\")","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:45.739294Z","iopub.execute_input":"2023-06-07T15:07:45.739647Z","iopub.status.idle":"2023-06-07T15:07:45.744421Z","shell.execute_reply.started":"2023-06-07T15:07:45.73962Z","shell.execute_reply":"2023-06-07T15:07:45.743243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axis = plt.subplots(3, 3, figsize = (9, 9))\nc = 0\nfor i in range(3):\n    for j in range(3):\n        # Start reading the files.\n        patient_id = labels.patientId.iloc[c]\n        \n        # Create a path to the dicom file of this particular patient/\n        dcm_path = ROOT_PATH/patient_id\n        \n        # Add .dcm extension\n        dcm_path = dcm_path.with_suffix(\".dcm\")\n        \n        # Reading the dicom file\n        dcm = pydicom.read_file(dcm_path).pixel_array\n        \n        # Extract the labels of the particular patient from labels dataframe\n        label = labels['Target'].iloc[c]\n        \n        # Visualize the image\n        axis[i][j].imshow(dcm, cmap = \"gray\")\n        axis[i][j].set_title(label)\n        \n        c += 1","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:46.85152Z","iopub.execute_input":"2023-06-07T15:07:46.851869Z","iopub.status.idle":"2023-06-07T15:07:49.278895Z","shell.execute_reply.started":"2023-06-07T15:07:46.851842Z","shell.execute_reply":"2023-06-07T15:07:49.277855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialization \nsums, sums_squared = 0, 0\n\n# Loop over all patient ids\n# To decide if a data is used as training data or validation data\n# We can use the enumerate \n\nfor c, patient_id, in enumerate(tqdm(labels.patientId)):\n    # Start reading the files.\n    patient_id = labels.patientId.iloc[c]\n        \n    # Create a path to the dicom file of this particular patient/\n    dcm_path = ROOT_PATH/patient_id\n\n    # Add .dcm extension\n    dcm_path = dcm_path.with_suffix(\".dcm\")\n\n    # Reading the dicom file and dividing the pixel array by 255\n    dcm = pydicom.read_file(dcm_path).pixel_array\n    \n    # Resizing the image and converting it's type\n    dcm_array = cv2.resize(dcm, (224, 224)).astype(np.float16)\n    \n    # Storing the label in 'label'\n    label = labels.Target.iloc[c]\n    \n    # Identify training/validation\n    train_or_val = \"train\" if c < 24000 else \"val\"\n    \n    # Save preprocessed images\n    current_save_path = SAVE_PATH/train_or_val/str(label)\n    current_save_path.mkdir(parents = True, exist_ok = True)\n    np.save(current_save_path/patient_id, dcm_array)\n    \n    \n    # Update sums and sums_squared\n    normalizer = 224*224","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:07:49.280424Z","iopub.execute_input":"2023-06-07T15:07:49.281418Z","iopub.status.idle":"2023-06-07T15:13:33.500056Z","shell.execute_reply.started":"2023-06-07T15:07:49.281384Z","shell.execute_reply":"2023-06-07T15:13:33.499121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = 0.49\nstd = 0.24","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:13:35.963056Z","iopub.execute_input":"2023-06-07T15:13:35.963716Z","iopub.status.idle":"2023-06-07T15:13:35.968699Z","shell.execute_reply.started":"2023-06-07T15:13:35.96368Z","shell.execute_reply":"2023-06-07T15:13:35.967283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:13:36.966625Z","iopub.execute_input":"2023-06-07T15:13:36.966972Z","iopub.status.idle":"2023-06-07T15:13:36.971716Z","shell.execute_reply.started":"2023-06-07T15:13:36.966944Z","shell.execute_reply":"2023-06-07T15:13:36.97062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transformation and affaine matrix for augmentaions\n\ntrain_transforms = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std),\n    transforms.RandomAffine(degrees = (-5, 5), translate = (0, 0.05), scale = (0.9, 1.1)),\n    transforms.RandomResizedCrop((224, 224), scale = (0.35, 1))\n])\n\nval_transforms = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:03.063107Z","iopub.execute_input":"2023-06-07T15:14:03.063497Z","iopub.status.idle":"2023-06-07T15:14:03.071319Z","shell.execute_reply.started":"2023-06-07T15:14:03.063468Z","shell.execute_reply":"2023-06-07T15:14:03.069441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DataLoader\n\ntrain_dataset = torchvision.datasets.DatasetFolder(\"/kaggle/working/Processed_images/train/\", loader = load_file, extensions = \"npy\", transform = train_transforms)\n\nval_dataset = torchvision.datasets.DatasetFolder(\"/kaggle/working/Processed_images/val/\", loader = load_file, extensions = \"npy\", transform = val_transforms)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:05.341842Z","iopub.execute_input":"2023-06-07T15:14:05.342207Z","iopub.status.idle":"2023-06-07T15:14:05.468149Z","shell.execute_reply.started":"2023-06-07T15:14:05.34216Z","shell.execute_reply":"2023-06-07T15:14:05.467174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\nnum_workers = 4","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:06.382813Z","iopub.execute_input":"2023-06-07T15:14:06.38317Z","iopub.status.idle":"2023-06-07T15:14:06.387675Z","shell.execute_reply.started":"2023-06-07T15:14:06.38314Z","shell.execute_reply":"2023-06-07T15:14:06.386398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:07.441225Z","iopub.execute_input":"2023-06-07T15:14:07.442118Z","iopub.status.idle":"2023-06-07T15:14:07.44713Z","shell.execute_reply.started":"2023-06-07T15:14:07.442078Z","shell.execute_reply":"2023-06-07T15:14:07.445969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataLoader(train_dataset, batch_size = batch_size, num_workers = num_workers, shuffle = True)\nval_loader = DataLoader(val_dataset, batch_size = batch_size, num_workers = num_workers, shuffle = False)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:08.277221Z","iopub.execute_input":"2023-06-07T15:14:08.278215Z","iopub.status.idle":"2023-06-07T15:14:08.28565Z","shell.execute_reply.started":"2023-06-07T15:14:08.278147Z","shell.execute_reply":"2023-06-07T15:14:08.284565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(train_dataset.targets, return_counts = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:27.392086Z","iopub.execute_input":"2023-06-07T15:14:27.392766Z","iopub.status.idle":"2023-06-07T15:14:27.402574Z","shell.execute_reply.started":"2023-06-07T15:14:27.392734Z","shell.execute_reply":"2023-06-07T15:14:27.401613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Observing the Resnet18 architecture\n\ntorchvision.models.resnet18()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:28.992083Z","iopub.execute_input":"2023-06-07T15:14:28.992754Z","iopub.status.idle":"2023-06-07T15:14:29.236477Z","shell.execute_reply.started":"2023-06-07T15:14:28.992722Z","shell.execute_reply":"2023-06-07T15:14:29.235405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PneumoniaModel(pl.LightningModule):\n    \n    # Constructor\n    def __init__(self):\n        super(PneumoniaModel, self).__init__()\n        self.model = torchvision.models.resnet18()\n        self.model.conv1 = torch.nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n        self.model.fc = torch.nn.Linear(in_features = 512, out_features = 1, bias = True)\n        \n        self.optimizer = torch.optim.Adam(self.model.parameters(), lr = 1e-4)\n        self.loss_fn = torch.nn.BCEWithLogitsLoss(pos_weight = torch.tensor([3]))\n        \n        self.train_acc = torchmetrics.Accuracy(task = \"binary\")\n        self.val_acc = torchmetrics.Accuracy(task = \"binary\")\n        #self.test_acc = torchmetrics.Accuracy()\n        \n        \n        \n    # Activation\n    def forward(self, data):\n        pred = self.model(data)\n        return pred\n    \n    def training_step(self, batch, batch_idx):\n        x_ray, label = batch\n        label = label.float()\n        pred = self(x_ray)[:, 0]\n        loss = self.loss_fn(pred, label)\n        \n        self.log(\"Train Loss\", loss)\n        self.log(\"Step Train ACC\", self.train_acc(torch.sigmoid(pred), label.int()))\n        \n        return loss\n    \n    \n    \n    def on_train_epoch_end(self):\n        self.log(\"Train ACC\", self.train_acc.compute())\n        \n    \n    def validation_step(self, batch, batch_idx):\n        x_ray, label = batch\n        label = label.float()\n        pred = self(x_ray)[:, 0]\n        loss = self.loss_fn(pred, label)\n        \n        self.log(\"Val Loss\", loss)\n        self.log(\"Step Val ACC\", self.val_acc(torch.sigmoid(pred), label.int()))\n\n        \n        \n       \n    def on_validation_epoch_end(self):\n        self.log(\"Val ACC\", self.val_acc.compute())\n        \n        \n    def configure_optimizers(self):\n        return[self.optimizer]","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:30.954313Z","iopub.execute_input":"2023-06-07T15:14:30.955262Z","iopub.status.idle":"2023-06-07T15:14:30.968466Z","shell.execute_reply.started":"2023-06-07T15:14:30.955222Z","shell.execute_reply":"2023-06-07T15:14:30.967506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = PneumoniaModel()\ncheckpoint_callback = ModelCheckpoint(\n    monitor = \"Val ACC\",\n    save_top_k = 1,\n    mode = \"max\",\n    dirpath = \"./saved_models\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:32.94017Z","iopub.execute_input":"2023-06-07T15:14:32.940567Z","iopub.status.idle":"2023-06-07T15:14:33.164921Z","shell.execute_reply.started":"2023-06-07T15:14:32.940538Z","shell.execute_reply":"2023-06-07T15:14:33.163962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configure the trainer\ntrainer = pl.Trainer(logger = TensorBoardLogger(save_dir = \"./logs\"), log_every_n_steps = 64, callbacks = checkpoint_callback, max_epochs = 2)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:34.525554Z","iopub.execute_input":"2023-06-07T15:14:34.525904Z","iopub.status.idle":"2023-06-07T15:14:35.383852Z","shell.execute_reply.started":"2023-06-07T15:14:34.525875Z","shell.execute_reply":"2023-06-07T15:14:35.382877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\ntrainer.fit(model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:14:36.748507Z","iopub.execute_input":"2023-06-07T15:14:36.748861Z","iopub.status.idle":"2023-06-07T15:17:06.029706Z","shell.execute_reply.started":"2023-06-07T15:14:36.748832Z","shell.execute_reply":"2023-06-07T15:17:06.028671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if GPU is available\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel = PneumoniaModel.load_from_checkpoint(\"/kaggle/working/saved_models/epoch=1-step=750.ckpt\")\nmodel.eval()\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:23:55.614387Z","iopub.execute_input":"2023-06-07T15:23:55.614776Z","iopub.status.idle":"2023-06-07T15:23:56.005855Z","shell.execute_reply.started":"2023-06-07T15:23:55.61474Z","shell.execute_reply":"2023-06-07T15:23:56.00476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls /kaggle/working/saved_models","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:23:36.220325Z","iopub.execute_input":"2023-06-07T15:23:36.220712Z","iopub.status.idle":"2023-06-07T15:23:37.322236Z","shell.execute_reply.started":"2023-06-07T15:23:36.220681Z","shell.execute_reply":"2023-06-07T15:23:37.320895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nlabels = []\n\nwith torch.no_grad():\n    for data, label in tqdm(val_dataset):\n        data = data.to(device).float().unsqueeze(0)\n        \n        # Calculate probabilities\n        pred = torch.sigmoid(model(data)[0].cpu())\n        \n        preds.append(pred)\n        labels.append(label)\npreds = torch.tensor(preds)\nlabels = torch.tensor(labels).int()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:24:04.208637Z","iopub.execute_input":"2023-06-07T15:24:04.209583Z","iopub.status.idle":"2023-06-07T15:24:14.629715Z","shell.execute_reply.started":"2023-06-07T15:24:04.209549Z","shell.execute_reply":"2023-06-07T15:24:14.628824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Accuracy\nacc = torchmetrics.Accuracy(task = \"binary\")(preds, labels)\n\n# Precision\nprecision = torchmetrics.Precision(task = \"binary\")(preds, labels)\n\n# Recall\nrecall = torchmetrics.Recall(task = \"binary\")(preds, labels)\n\n# Confusion Matrix\ncm = torchmetrics.ConfusionMatrix(num_classes=2, task = \"binary\")(preds, labels)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:25:21.207105Z","iopub.execute_input":"2023-06-07T15:25:21.208133Z","iopub.status.idle":"2023-06-07T15:25:21.232345Z","shell.execute_reply.started":"2023-06-07T15:25:21.208099Z","shell.execute_reply":"2023-06-07T15:25:21.231425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Accuracy : {acc}\")\nprint(f\"Precision : {precision}\")\nprint(f\"Recall : {recall}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-07T15:25:35.598204Z","iopub.execute_input":"2023-06-07T15:25:35.599315Z","iopub.status.idle":"2023-06-07T15:25:35.605718Z","shell.execute_reply.started":"2023-06-07T15:25:35.599212Z","shell.execute_reply":"2023-06-07T15:25:35.604436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}