{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport pydicom\nimport numpy as np\nimport cv2\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:05.067159Z","iopub.execute_input":"2023-03-10T11:49:05.068069Z","iopub.status.idle":"2023-03-10T11:49:05.631071Z","shell.execute_reply.started":"2023-03-10T11:49:05.067942Z","shell.execute_reply":"2023-03-10T11:49:05.629678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:05.633241Z","iopub.execute_input":"2023-03-10T11:49:05.633707Z","iopub.status.idle":"2023-03-10T11:49:05.727525Z","shell.execute_reply.started":"2023-03-10T11:49:05.633675Z","shell.execute_reply":"2023-03-10T11:49:05.726149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:05.729195Z","iopub.execute_input":"2023-03-10T11:49:05.730131Z","iopub.status.idle":"2023-03-10T11:49:05.756196Z","shell.execute_reply.started":"2023-03-10T11:49:05.730074Z","shell.execute_reply":"2023-03-10T11:49:05.754727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = labels.drop_duplicates(\"patientId\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:05.759944Z","iopub.execute_input":"2023-03-10T11:49:05.761145Z","iopub.status.idle":"2023-03-10T11:49:05.788682Z","shell.execute_reply.started":"2023-03-10T11:49:05.761077Z","shell.execute_reply":"2023-03-10T11:49:05.787198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_PATH = Path(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images\")\nSAVE_PATH = Path(\"/kaggle/working/Processed\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:05.790314Z","iopub.execute_input":"2023-03-10T11:49:05.790727Z","iopub.status.idle":"2023-03-10T11:49:05.797488Z","shell.execute_reply.started":"2023-03-10T11:49:05.790688Z","shell.execute_reply":"2023-03-10T11:49:05.796076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axis = plt.subplots(4, 4, figsize=(9, 9))\nc = 0\nfor i in range(4):\n    for j in range(4):\n        patient_id = labels.patientId.iloc[c]\n        dcm_path = ROOT_PATH/patient_id\n        dcm_path = dcm_path.with_suffix(\".dcm\")\n        dcm = pydicom.read_file(dcm_path).pixel_array\n        \n        label = labels[\"Target\"].iloc[c]\n        \n        axis[i][j].imshow(dcm, cmap=\"bone\")\n        axis[i][j].set_title(label)\n        c+=1","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:05.799078Z","iopub.execute_input":"2023-03-10T11:49:05.799518Z","iopub.status.idle":"2023-03-10T11:49:10.933129Z","shell.execute_reply.started":"2023-03-10T11:49:05.799481Z","shell.execute_reply":"2023-03-10T11:49:10.931955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sums = 0\nsums_squared = 0\n\nfor c, patient_id in enumerate(tqdm(labels.patientId)):\n    dcm_path = ROOT_PATH/patient_id  # Create the path to the dcm file\n    dcm_path = dcm_path.with_suffix(\".dcm\")  # And add the .dcm suffix\n    \n    # Read the dicom file with pydicom and standardize the array\n    dcm = pydicom.read_file(dcm_path).pixel_array / 255  \n        \n    # Resize the image as 1024x1024 is way to large to be handeled by Deep Learning models at the moment\n    # Let's use a shape of 224x224\n    # In order to use less space when storing the image we convert it to float16\n    dcm_array = cv2.resize(dcm, (224, 224)).astype(np.float16)\n    \n    # Retrieve the corresponding label\n    label = labels.Target.iloc[c]\n    \n    # 4/5 train split, 1/5 val split\n    train_or_val = \"train\" if c < 24000 else \"val\" \n        \n    current_save_path = SAVE_PATH/train_or_val/str(label) # Define save path and create if necessary\n    current_save_path.mkdir(parents=True, exist_ok=True)\n    np.save(current_save_path/patient_id, dcm_array)  # Save the array in the corresponding directory\n    \n    normalizer = dcm_array.shape[0] * dcm_array.shape[1]  # Normalize sum of image\n    if train_or_val == \"train\":  # Only use train data to compute dataset statistics\n        sums += np.sum(dcm_array) / normalizer\n        sums_squared += (np.power(dcm_array, 2).sum()) / normalizer\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T11:49:10.93454Z","iopub.execute_input":"2023-03-10T11:49:10.935081Z","iopub.status.idle":"2023-03-10T12:04:11.02525Z","shell.execute_reply.started":"2023-03-10T11:49:10.935037Z","shell.execute_reply":"2023-03-10T12:04:11.023032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = sums / 24000\nstd = np.sqrt(sums_squared / 24000 - (mean**2))","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:11.02861Z","iopub.execute_input":"2023-03-10T12:04:11.029271Z","iopub.status.idle":"2023-03-10T12:04:11.037243Z","shell.execute_reply.started":"2023-03-10T12:04:11.02921Z","shell.execute_reply":"2023-03-10T12:04:11.035744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Mean of Dataset: {mean}, STD: {std}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:11.039299Z","iopub.execute_input":"2023-03-10T12:04:11.040064Z","iopub.status.idle":"2023-03-10T12:04:11.053793Z","shell.execute_reply.started":"2023-03-10T12:04:11.040025Z","shell.execute_reply":"2023-03-10T12:04:11.05227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torchvision\nfrom torchvision import transforms\nimport torchmetrics\nimport pytorch_lightning as pl\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom pytorch_lightning.loggers import TensorBoardLogger","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:11.057845Z","iopub.execute_input":"2023-03-10T12:04:11.058239Z","iopub.status.idle":"2023-03-10T12:04:17.742049Z","shell.execute_reply.started":"2023-03-10T12:04:11.058206Z","shell.execute_reply":"2023-03-10T12:04:17.740436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:17.744266Z","iopub.execute_input":"2023-03-10T12:04:17.745027Z","iopub.status.idle":"2023-03-10T12:04:17.750741Z","shell.execute_reply.started":"2023-03-10T12:04:17.744987Z","shell.execute_reply":"2023-03-10T12:04:17.749406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_transforms = transforms.Compose([\n                                    transforms.ToTensor(),  # Convert numpy array to tensor\n                                    transforms.Normalize(0.49, 0.248),  # Use mean and std from preprocessing notebook\n                                    transforms.RandomAffine( # Data Augmentation\n                                        degrees=(-5, 5), translate=(0, 0.05), scale=(0.9, 1.1)),\n                                        transforms.RandomResizedCrop((224, 224), scale=(0.35, 1))\n\n])\n\nval_transforms = transforms.Compose([\n                                    transforms.ToTensor(),  # Convert numpy array to tensor\n                                    transforms.Normalize([0.49], [0.248]),  # Use mean and std from preprocessing notebook\n])\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:17.752777Z","iopub.execute_input":"2023-03-10T12:04:17.753657Z","iopub.status.idle":"2023-03-10T12:04:17.764436Z","shell.execute_reply.started":"2023-03-10T12:04:17.753609Z","shell.execute_reply":"2023-03-10T12:04:17.763211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = torchvision.datasets.DatasetFolder(\n    \"/kaggle/working/Processed/train\",\n    loader=load_file, extensions=\"npy\", transform=train_transforms)\n\nval_dataset = torchvision.datasets.DatasetFolder(\n    \"/kaggle/working/Processed/val\",\n    loader=load_file, extensions=\"npy\", transform=val_transforms)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:17.766226Z","iopub.execute_input":"2023-03-10T12:04:17.76673Z","iopub.status.idle":"2023-03-10T12:04:17.914887Z","shell.execute_reply.started":"2023-03-10T12:04:17.766671Z","shell.execute_reply":"2023-03-10T12:04:17.913527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axis = plt.subplots(2, 2, figsize=(9, 9))\nfor i in range(2):\n    for j in range(2):\n        random_index = np.random.randint(0, 20000)\n        x_ray, label = train_dataset[random_index]\n        axis[i][j].imshow(x_ray[0], cmap=\"bone\")\n        axis[i][j].set_title(f\"Label:{label}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:17.916701Z","iopub.execute_input":"2023-03-10T12:04:17.917099Z","iopub.status.idle":"2023-03-10T12:04:18.798248Z","shell.execute_reply.started":"2023-03-10T12:04:17.917063Z","shell.execute_reply":"2023-03-10T12:04:18.797246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64#TODO\nnum_workers = 4# TODO\n\ntrain_loader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size, num_workers=num_workers, shuffle=True)\nval_loader = torch.utils.data.DataLoader(val_dataset, batch_size=batch_size, num_workers=num_workers, shuffle=False)\n\nprint(f\"There are {len(train_dataset)} train images and {len(val_dataset)} val images\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:18.799513Z","iopub.execute_input":"2023-03-10T12:04:18.800634Z","iopub.status.idle":"2023-03-10T12:04:18.808481Z","shell.execute_reply.started":"2023-03-10T12:04:18.800593Z","shell.execute_reply":"2023-03-10T12:04:18.807311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(train_dataset.targets, return_counts=True), np.unique(val_dataset.targets, return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:18.810024Z","iopub.execute_input":"2023-03-10T12:04:18.81079Z","iopub.status.idle":"2023-03-10T12:04:18.829885Z","shell.execute_reply.started":"2023-03-10T12:04:18.810735Z","shell.execute_reply":"2023-03-10T12:04:18.828404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torchvision.models.resnet18()","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:04:18.831786Z","iopub.execute_input":"2023-03-10T12:04:18.832408Z","iopub.status.idle":"2023-03-10T12:04:19.085826Z","shell.execute_reply.started":"2023-03-10T12:04:18.83237Z","shell.execute_reply":"2023-03-10T12:04:19.084576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PneumoniaModel(pl.LightningModule):\n    def __init__(self, weight=1):\n        super().__init__()\n        \n        self.model = torchvision.models.resnet18()\n        # change conv1 from 3 to 1 input channels\n        self.model.conv1 = torch.nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n        # change out_feature of the last fully connected layer (called fc in resnet18) from 1000 to 1\n        self.model.fc = torch.nn.Linear(in_features=512, out_features=1)\n        \n        self.optimizer = torch.optim.Adam(self.model.parameters(), lr=1e-4)\n        self.loss_fn = torch.nn.BCEWithLogitsLoss(pos_weight=torch.tensor([weight]))\n        \n        # simple accuracy computation\n        self.train_acc = torchmetrics.Accuracy()\n        self.val_acc = torchmetrics.Accuracy()\n\n    def forward(self, data):\n        pred = self.model(data)\n        return pred\n    \n    def training_step(self, batch, batch_idx):\n        x_ray, label = batch\n        label = label.float()  # Convert label to float (just needed for loss computation)\n        pred = self(x_ray)[:,0]  # Prediction: Make sure prediction and label have same shape\n        loss = self.loss_fn(pred, label)  # Compute the loss\n        \n        # Log loss and batch accuracy\n        self.log(\"Train Loss\", loss)\n        self.log(\"Step Train Acc\", self.train_acc(torch.sigmoid(pred), label.int()))\n        return loss\n    \n    \n    def training_epoch_end(self, outs):\n        # After one epoch compute the whole train_data accuracy\n        self.log(\"Train Acc\", self.train_acc.compute())\n        \n        \n    def validation_step(self, batch, batch_idx):\n        # Same steps as in the training_step\n        x_ray, label = batch\n        label = label.float()\n        pred = self(x_ray)[:,0]  # make sure prediction and label have same shape\n\n        loss = self.loss_fn(pred, label)\n        \n        # Log validation metrics\n        self.log(\"Val Loss\", loss)\n        self.log(\"Step Val Acc\", self.val_acc(torch.sigmoid(pred), label.int()))\n        return loss\n    \n    def validation_epoch_end(self, outs):\n        self.log(\"Val Acc\", self.val_acc.compute())\n    \n    def configure_optimizers(self):\n        #Caution! You always need to return a list here (just pack your optimizer into one :))\n        return [self.optimizer]\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:06:43.440627Z","iopub.execute_input":"2023-03-10T12:06:43.441518Z","iopub.status.idle":"2023-03-10T12:06:43.459755Z","shell.execute_reply.started":"2023-03-10T12:06:43.441465Z","shell.execute_reply":"2023-03-10T12:06:43.458223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = PneumoniaModel()  # Instanciate the model","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:07:05.740691Z","iopub.execute_input":"2023-03-10T12:07:05.741207Z","iopub.status.idle":"2023-03-10T12:07:06.053225Z","shell.execute_reply.started":"2023-03-10T12:07:05.741166Z","shell.execute_reply":"2023-03-10T12:07:06.051297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the checkpoint callback\ncheckpoint_callback = ModelCheckpoint(\n    monitor='Val Acc',\n    save_top_k=10,\n    mode='max')","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:07:13.39147Z","iopub.execute_input":"2023-03-10T12:07:13.391929Z","iopub.status.idle":"2023-03-10T12:07:13.399535Z","shell.execute_reply.started":"2023-03-10T12:07:13.391893Z","shell.execute_reply":"2023-03-10T12:07:13.398033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the trainer\n# Change the gpus parameter to the number of available gpus on your system. Use 0 for CPU training\n\ngpus = 0 #TODO\ntrainer = pl.Trainer(gpus=gpus, logger=TensorBoardLogger(save_dir=\"./logs\"), log_every_n_steps=1,\n                     callbacks=checkpoint_callback,\n                     max_epochs=35)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:07:21.479384Z","iopub.execute_input":"2023-03-10T12:07:21.479824Z","iopub.status.idle":"2023-03-10T12:07:21.700103Z","shell.execute_reply.started":"2023-03-10T12:07:21.479785Z","shell.execute_reply":"2023-03-10T12:07:21.698868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.fit(model, train_loader, val_loader)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T12:07:44.929749Z","iopub.execute_input":"2023-03-10T12:07:44.930254Z","iopub.status.idle":"2023-03-10T12:07:44.963266Z","shell.execute_reply.started":"2023-03-10T12:07:44.930215Z","shell.execute_reply":"2023-03-10T12:07:44.961618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n\n# Use strict=False, otherwise we would want to match the pos_weight which is not necessary\nmodel = PneumoniaModel.load_from_checkpoint(\"weights/weights_1.ckpt\")\nmodel.eval()\nmodel.to(device);\n\npreds = []\nlabels = []\n\nwith torch.no_grad():\n    for data, label in tqdm(val_dataset):\n        data = data.to(device).float().unsqueeze(0)\n        pred = torch.sigmoid(model(data)[0].cpu())\n        preds.append(pred)\n        labels.append(label)\npreds = torch.tensor(preds)\nlabels = torch.tensor(labels).int()\n\nacc = torchmetrics.Accuracy()(preds, labels)\nprecision = torchmetrics.Precision()(preds, labels)\nrecall = torchmetrics.Recall()(preds, labels)\ncm = torchmetrics.ConfusionMatrix(num_classes=2)(preds, labels)\ncm_threshed = torchmetrics.ConfusionMatrix(num_classes=2, threshold=0.25)(preds, labels)\n\nprint(f\"Val Accuracy: {acc}\")\nprint(f\"Val Precision: {precision}\")\nprint(f\"Val Recall: {recall}\")\nprint(f\"Confusion Matrix:\\n {cm}\")\nprint(f\"Confusion Matrix 2:\\n {cm_threshed}\")","metadata":{},"execution_count":null,"outputs":[]}]}