# %% [markdown]
# # About 
# 
# The idea is discussed [here, method 1](https://www.kaggle.com/c/siim-covid19-detection/discussion/245323), that is to **add an extras supervison** to the classificaiton mdoel by adding the **segmentation loss**. Basically, we make the use of the bounding box information to crop the anomaly region from the chest x-ray samples and further use those cropped samples as an additional targets for the classifier model at **cetain layer**. Here is the general overview of this approach. 
# 
# ![Untitled-2](https://user-images.githubusercontent.com/17668390/122642346-dc044b80-d12b-11eb-9cb9-d1771a682541.png)
# 
# <div align="center">
#   Figure 1: Cropped ROI Segment for the Additional Supervision to the Classifier.
# </div>

# %% [markdown]
# For input image of shape `512`, we will get `16` dimention of `n` depth feature maps at the last layer of base model. As the segmentation target would be one channel and downsampled to match last layer feature maps, we need to add a simple **mask head**. 

# %% [code] {"execution":{"iopub.status.busy":"2021-06-29T07:43:14.184096Z","iopub.execute_input":"2021-06-29T07:43:14.184455Z","iopub.status.idle":"2021-06-29T07:43:19.380494Z","shell.execute_reply.started":"2021-06-29T07:43:14.18438Z","shell.execute_reply":"2021-06-29T07:43:19.378978Z"}}
import os, math
import psutil, random 

import numpy as np 
import pandas as pd 
import matplotlib.pyplot as plt

import cv2; print(cv2.__version__)
import tensorflow as tf; print(tf.__version__)

# %% [code]
MIXED_PRECISION = True
XLA_ACCELERATE  = False

GPUS = tf.config.experimental.list_physical_devices('GPU')
if GPUS:
    try:
        for GPU in GPUS:
            tf.config.experimental.set_memory_growth(GPU, True)
            logical_gpus = tf.config.experimental.list_logical_devices('GPU')
            print(len(GPUS), "Physical GPUs,", len(logical_gpus), "Logical GPUs") 
    except RuntimeError as  RE:
        print(RE)

if MIXED_PRECISION:
    policy = tf.keras.mixed_precision.experimental.Policy('mixed_float16')
    tf.keras.mixed_precision.experimental.set_policy(policy)
    print('Mixed precision enabled')

if XLA_ACCELERATE:
    tf.config.optimizer.set_jit(True)
    print('Accelerated Linear Algebra enabled')
    
print("Tensorflow version " + tf.__version__)

# %% [markdown]
# # Data Preprocess

# %% [code]
study_df = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv'); print(study_df.shape)
study_df['StudyInstanceUID'] = study_df['id'].apply(lambda x: x.replace('_study', ''))
del study_df['id']

def hot_to_sparse(row):
    return(row.index[row.apply(lambda x: x==1)][0])
study_df['diagnosis'] = study_df.apply(lambda row:hot_to_sparse(row), axis=1)
cls = {
    'Typical Appearance':1,                    
    'Negative for Pneumonia':2,                
    'Indeterminate Appearance':3,                     
    'Atypical Appearance':4,    
}
study_df['sparse_gt'] = study_df.diagnosis.map(cls) 

image_df = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv'); print(image_df.shape)
train = image_df.merge(study_df, on='StudyInstanceUID')
train['id'] = train['id'].apply(lambda x: x.replace('_image', ''))
display(train.head()); print(train.shape)

# %% [markdown]
# # ROI Segment: Cropped Bounding Box

# %% [code]
def vis(path1, path2, n_images, is_random=True, figsize=(16, 16)):
    '''
    https://gist.github.com/innat/00de7561033ba373745d425c6da7bf8c
    '''
    image_names = os.listdir(path1)
    masks_names = os.listdir(path2)
    
    for i in range(n_images):
        if is_random:
            image_name = random.choice(masks_names)
            masks_name = image_name
        else:
            image_name = masks_names[i]
            masks_name = masks_names[i]
            
        img = cv2.resize(cv2.imread(os.path.join(path1, image_name)), (512, 512))
        msk = cv2.resize(cv2.imread(os.path.join(path2, masks_name)), (512, 512))
        
        plt.figure(figsize=(20,20))
        plt.subplot(121); plt.imshow(img);
        plt.subplot(122); plt.imshow(msk);
        plt.show()

# %% [code]
base_path = '../input/covid19-detection-890pxpng-study'
TRAIN_IMG_PATH =  os.path.join(base_path, 'train/')
TRAIN_MSK_PATH = os.path.join(base_path, 'ROI Mask/')

vis(TRAIN_IMG_PATH, TRAIN_MSK_PATH, 5, is_random=True)

# %% [code]
from sklearn.model_selection import StratifiedKFold

skf = StratifiedKFold(n_splits=3, shuffle=True, random_state=101)
for index, (train_index, val_index) in enumerate(skf.split(X=train.index, y=train.sparse_gt)):
    train.loc[val_index, 'fold'] = index
    
print(train.groupby(['fold', train.sparse_gt]).size())

# %% [markdown]
# # Data Generator

# %% [code]
import albumentations as A 

# For Validation 
def albu_transforms_train(data_resize): 
    return A.Compose([
        A.Resize(data_resize, data_resize)
        #A.ToFloat(), # no need if use keras.applicaiton.EfficientNets Bx, built-in normalization 
    ], p=1.)


# For Validation 
def albu_transforms_valid(data_resize): 
    return A.Compose([
        A.Resize(data_resize, data_resize)
        #A.ToFloat(), # no need if use keras.applicaiton.EfficientNets Bx
        ], p=1.)

# %% [code]
class Covid19Generator(tf.keras.utils.Sequence):
    def __init__(self, img_path, msk_path, data, batch_size, random_state, 
                 idim, mdim, shuffle=True, transform=None, is_train=False):
        self.idim = idim
        self.mdim = mdim  
        self.data = data
        self.shuffle  = shuffle
        self.random_state = random_state
        
        self.img_path = img_path
        self.msk_path = msk_path
        self.is_train = is_train
        
        self.augment  = transform
        self.batch_size = batch_size
        
        self.list_idx = data.index.values
        self.label = self.data[['Negative for Pneumonia', 
                                'Typical Appearance', 
                                'Indeterminate Appearance', 
                                'Atypical Appearance']] if self.is_train else np.nan
        self.on_epoch_end()
        
    def __len__(self):
        return int(np.floor(len(self.list_idx) / self.batch_size))
    
    def __getitem__(self, index):
        batch_idx = self.indices[index*self.batch_size:(index+1)*self.batch_size]
        idx = [self.list_idx[k] for k in batch_idx]
        
        Data = np.zeros((self.batch_size,) + self.idim + (3,), dtype="float32")
        Mask = np.zeros((self.batch_size,) + self.mdim + (1,), dtype="float32")
        Target = np.zeros((self.batch_size, 4), dtype = np.float32)

        for i, k in enumerate(idx):
            # load the image file using cv2
            image = cv2.imread(self.img_path + self.data['id'][k] + '.png')[:, :, [2, 1, 0]]
            mask = cv2.imread(self.msk_path + self.data['id'][k] + '.png', 0)
            
            try:
                mask = cv2.resize(mask, self.mdim)[:, :, np.newaxis]
            except:
                mask = np.zeros_like(cv2.resize(image[:,:,:1], self.mdim))[:, :, np.newaxis]
          
            res = self.augment(image=image)
            image = res['image']
            
            # mask normalization must
            mask = mask.astype(np.float32)/255.0 

            # assign 
            if self.is_train:
                Data[i,] = image
                Mask[i,] = mask
                Target[i,] = self.label.iloc[k,].values #.values
            else:
                Data[i,] =  image 
        
        inps = {'input': Data}
        outs = {'clss': Target, 'segg': Mask}
        return inps, outs
    
    def on_epoch_end(self):
        self.indices = np.arange(len(self.list_idx))
        if self.shuffle:
            np.random.seed(self.random_state)
            np.random.shuffle(self.indices)

# %% [code]
import matplotlib.pyplot as plt 
from pylab import rcParams

# helper function to plot sample 
def plot_imgs(dataset_show, row, col):
    rcParams['figure.figsize'] = 20,10
    for i in range(row):
        f, ax = plt.subplots(1,col)
        for p in range(col):
            idx = np.random.randint(0, len(dataset_show))
            img, label = dataset_show[idx]
            ax[p].grid(False)
            ax[p].imshow(label['segg'][0], cmap='gray')
            ax[p].set_title(label['clss'][0])
    plt.show()

# %% [code]
fold = 0
img_size = 512
msk_sizze = 16
batch_size = 32

def count_data_items(length, b_max):
    batch_size = sorted([int(length/n) for n in range(1, length+1) \
                         if length % n == 0 and length/n <= b_max], reverse=True)[0]  
    steps  = length / batch_size 
    return batch_size, steps

def fold_generator(fold):
    # for way one - data generator
    train_labels = train[train.fold != fold].reset_index(drop=True)
    val_labels = train[train.fold == fold].reset_index(drop=True)

    train_generator = Covid19Generator(TRAIN_IMG_PATH, TRAIN_MSK_PATH,
                              train_labels, 
                              batch_size, 1234, (img_size, img_size), (msk_sizze, msk_sizze),
                              shuffle = True, is_train = True,
                              transform = albu_transforms_train(img_size))
    
    valid_batch, valid_step = count_data_items(len(val_labels), batch_size)

    val_generator = Covid19Generator(TRAIN_IMG_PATH, TRAIN_MSK_PATH,
                              val_labels, 
                              valid_batch, 1234, (img_size, img_size), (msk_sizze, msk_sizze),
                              shuffle = False, is_train = True,
                              transform = albu_transforms_valid(img_size))

    return train_generator, val_generator, train_labels, val_labels, valid_step


# first fold 
train_gen, val_gen, train_len, val_len, val_step = fold_generator(fold)

# %% [code]
plot_imgs(train_gen, 5, 4) # plotting only 16x mask

# %% [markdown]
# # Model 
# 
# For demonstration purpose, we pick `EfficientNet` model with image shape `512 x 512`. And for segmentation prediction maps, we will pick layer `top_activation` and pass it to **maks head** branch further processsing. 
# 
# ```python
# for layer in applications.EfficientNetB1(input_shape=(512, 512, 3), weights=None).layers[-10:]:
#     print(layer.name, layer.output_shape)
#     
# [out]:
# block7b_project_conv (None, 16, 16, 320)
# block7b_project_bn (None, 16, 16, 320)
# block7b_drop (None, 16, 16, 320)
# block7b_add (None, 16, 16, 320)
# top_conv (None, 16, 16, 1280)
# top_bn (None, 16, 16, 1280)
# top_activation (None, 16, 16, 1280)
# 
# # moved out for top free
# avg_pool (None, 1280)
# top_dropout (None, 1280)
# predictions (None, 1000)
# ```

# %% [code]
from tensorflow.keras import Model 
from tensorflow.keras import Sequential 
from tensorflow.keras import Input 
from tensorflow.keras import layers 
from tensorflow.keras import applications 

class CovidNet(Model):
    def __init__(self):
        super(CovidNet, self).__init__()
        self.base = applications.EfficientNetB1(input_shape=(512, 512, 3),
                                                  include_top=False,
                                                  weights='imagenet')
        # desired model 
        self.base = Model(
                [self.base.inputs], 
                [self.base.get_layer('top_activation').output, self.base.output]
            )
        
        # tail / head for the classifier 
        self.tail = Sequential(
            [
                layers.GlobalAveragePooling2D(),
                layers.Dropout(0.2),
                layers.BatchNormalization(),
                layers.Dense(4),
                layers.Softmax()
            ]
        )
        
        # tail / head for the mask 
        self.msk = Sequential(
            [
                layers.Conv2D(filters=512, kernel_size=(1, 1), strides=(1, 1), padding="same"),
                layers.ReLU(),
                layers.BatchNormalization(),
                layers.Conv2D(filters=1, kernel_size=(1,1), padding="same")
            ]
        )

    # feed-forwarding  
    def call(self, inputs, training=None, **kwargs):
        segg, clss = self.base(inputs['input'])

        return {
            'clss': self.tail(clss), 
            'segg': self.msk(segg)
        }
    

tf.keras.backend.clear_session()
model = CovidNet()
model.build(input_shape={'input': (None, 512, 512, 3)})
model.summary()

# %% [markdown]
# # Training

# %% [code]
# https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/schedules
from tensorflow.keras.optimizers.schedules import LearningRateSchedule, ExponentialDecay

class WarmupLearningRateSchedule(LearningRateSchedule):
    """Provides a variety of learning rate decay schedules with warm up."""

    def __init__(self,
               initial_lr,
               steps_per_epoch=None,
               lr_decay_type='exponential',
               decay_factor=0.97,
               decay_epochs=2.4,
               total_steps=None,
               warmup_epochs=5,
               minimal_lr=0):
        super(WarmupLearningRateSchedule, self).__init__()
        self.initial_lr = initial_lr
        self.steps_per_epoch = steps_per_epoch
        self.lr_decay_type = lr_decay_type
        self.decay_factor = decay_factor
        self.decay_epochs = decay_epochs
        self.total_steps = total_steps
        self.warmup_epochs = warmup_epochs
        self.minimal_lr = minimal_lr

    def __call__(self, step):
        if self.lr_decay_type == 'exponential':
            assert self.steps_per_epoch is not None
            decay_steps = self.steps_per_epoch * self.decay_epochs
            lr = ExponentialDecay(self.initial_lr, decay_steps, 
                                  self.decay_factor, staircase=True)(step)
        elif self.lr_decay_type == 'cosine':
            assert self.total_steps is not None
            lr = 0.5 * self.initial_lr * (
              1 + tf.cos(np.pi * tf.cast(step, tf.float32) / self.total_steps))
            
        elif self.lr_decay_type == 'linear':
            assert self.total_steps is not None
            lr = (1.0 - tf.cast(step, tf.float32) / self.total_steps) * self.initial_lr
        elif self.lr_decay_type == 'constant':
            lr = self.initial_lr
        else:
            assert False, 'Unknown lr_decay_type : %s' % self.lr_decay_type

        if self.minimal_lr:
            lr = tf.math.maximum(lr, self.minimal_lr)

        if self.warmup_epochs:
            warmup_steps = int(self.warmup_epochs * self.steps_per_epoch)
            warmup_lr = (
              self.initial_lr * tf.cast(step, tf.float32) /
              tf.cast(warmup_steps, tf.float32))
            lr = tf.cond(step < warmup_steps, lambda: warmup_lr, lambda: lr)

        return lr

    def get_config(self):
        return {
            'initial_lr': self.initial_lr,
            'steps_per_epoch': self.steps_per_epoch,
            'lr_decay_type': self.lr_decay_type,
            'decay_factor': self.decay_factor,
            'decay_epochs': self.decay_epochs,
            'total_steps': self.total_steps,
            'warmup_epochs': self.warmup_epochs,
            'minimal_lr': self.minimal_lr,
        }

# %% [code]
steps_per_epoch  = np.ceil(float(len(train_len)) / batch_size) 
validation_steps = val_step 
epochs = 3

lr_sched = 'cosine'
lr_base = 0.016
lr_min=0
lr_decay_epoch = 2.4
lr_warmup_epoch = 5
lr_decay_factor = 0.97

scaled_lr = lr_base * (batch_size / 256.0)
scaled_lr_min = lr_min * (batch_size / 256.0)
total_steps = steps_per_epoch * epochs

learning_rate = WarmupLearningRateSchedule(
    scaled_lr,
    steps_per_epoch=steps_per_epoch,
    decay_epochs=lr_decay_epoch,
    warmup_epochs=lr_warmup_epoch,
    decay_factor=lr_decay_factor,
    lr_decay_type=lr_sched,
    total_steps=total_steps,
    minimal_lr=scaled_lr_min)

# %% [code]
from tensorflow.keras import losses 
from tensorflow.keras import metrics
from tensorflow.keras import optimizers

# bind all
model.compile(
    loss = {
        'clss': losses.CategoricalCrossentropy(
            label_smoothing=0, from_logits=False),
        'segg': losses.BinaryCrossentropy(from_logits=True)
    },
    
    metrics = {
        'clss': [
            metrics.AUC(curve='ROC', multi_label=True),
            metrics.SpecificityAtSensitivity(0.60, name='@sensitivity')
        ]
    },
    
    optimizer = optimizers.Adam(learning_rate)
)

# list of call backs 
from tensorflow.keras import callbacks
callback_list = [
       callbacks.ModelCheckpoint(
            filepath='model.{epoch:02d}-{val_loss:.4f}.h5', 
            save_freq='epoch', verbose=1, monitor='val_loss', 
            save_weights_only=True, save_best_only=True
       )         
]

#https://www.kaggleusercontent.com/kf/66225692/eyJhbGciOiJkaXIiLCJlbmMiOiJBMTI4Q0JDLUhTMjU2In0..MHnndRGe8HGwAz9oNX5iqg.YWwci5bnnLcZsm29IKPMs1Byz6J5W_HHBbcfF2jOOYq1xGf9j0K5rJShm2kUKt_pgCr_n5tczipgpsZeJEhdSpE98bMVQBYX2sZ9NKwt-AQQ3R4IKw8exDLB6_wQ8qDxxWct06KaOP-Zwzptmsk8yQHAA2ww7m5cc--kk1LVo1XINoulbkgQdlUyITP3d4xBm6QZ4jifV-JoeFcxyUrAdTgh-kW548sGm5aT4rZ3CDQuv5_DgFbgDD2sIoeBVftLtx_gVXDHuRG9lIgjwMQlrEsuI-fg-20qgKJ5cN8t6HpNYBNPkPJRYSDgAPpBJu2zk71g8I3n_aq6kB2A5frcPJTBmEjTFrUcKcPzdrxdnLq7zJMgmk1xc7tABP8y8u1-Ik30k8Vm0YtClsNjRjF65aVf8GRoFw_OqkHACs2bOyTnMmtbaXYeFuIgaZxuJO8UvAjYvMMrM-L32iez6OsS-wciShe0JXxatMsxcPYZ8ylONIX3tRZJ0sMCVSBTgKz0ZLvTYDWZy_WPuKXvmPWLNdatNvjGCd-koYnEQJ2hYUUgYEZeybl72YfNG_5jpxcYwx7eeksp1rCVzlyMdqsLINtb0hzD1sWCe1tJ0YZEJMVjySyUSiuDqe2_Q9QxGePDxTmVj8pSaqQ_5gwck_KEd2Zy32eT-VW9EVCpkyWV2pC1yyrFCzNL2AOotxAmRmwk.648-AIpnkhuCfNcUr12aZQ/model.03-1.3574.h5
#Epoch 1/3
#132/132 - 293s - loss: 2.1199 - clss_loss: 1.4717 - segg_loss: 0.6482 - clss_auc: 0.6327 - clss_@sensitivity: 0.6883 - val_loss: 1.6351 - val_clss_loss: 1.1735 - val_segg_loss: 0.4616 - val_clss_auc: 0.6644 - val_clss_@sensitivity: 0.8204
#
#Epoch 00001: val_loss improved from inf to 1.63506, saving model to model.01-1.6351.h5
#Epoch 2/3
#132/132 - 248s - loss: 1.2958 - clss_loss: 1.0322 - segg_loss: 0.2637 - clss_auc: 0.7573 - clss_@sensitivity: 0.8893 - val_loss: 2.7119 - val_clss_loss: 2.4218 - val_segg_loss: 0.2903 - val_clss_auc: 0.5554 - val_clss_@sensitivity: 0.7088
#
#Epoch 00002: val_loss did not improve from 1.63506
#Epoch 3/3
#132/132 - 249s - loss: 1.2076 - clss_loss: 1.0339 - segg_loss: 0.1737 - clss_auc: 0.7563 - clss_@sensitivity: 0.8938 - val_loss: 1.3574 - val_clss_loss: 1.1735 - val_segg_loss: 0.1839 - val_clss_auc: 0.7064 - val_clss_@sensitivity: 0.7481
#
#Epoch 00003: val_loss improved from 1.63506 to 1.35738, saving model to model.03-1.3574.h5


# fitter 
if False:
    model.fit(train_gen, 
              steps_per_epoch=steps_per_epoch,
              validation_data=val_gen, 
              validation_steps=validation_steps,
              callbacks=callback_list, 
              workers=psutil.cpu_count(), verbose=1,
              epochs=epochs)

!wget -O model.h5 'https://www.kaggleusercontent.com/kf/66225692/eyJhbGciOiJkaXIiLCJlbmMiOiJBMTI4Q0JDLUhTMjU2In0..MHnndRGe8HGwAz9oNX5iqg.YWwci5bnnLcZsm29IKPMs1Byz6J5W_HHBbcfF2jOOYq1xGf9j0K5rJShm2kUKt_pgCr_n5tczipgpsZeJEhdSpE98bMVQBYX2sZ9NKwt-AQQ3R4IKw8exDLB6_wQ8qDxxWct06KaOP-Zwzptmsk8yQHAA2ww7m5cc--kk1LVo1XINoulbkgQdlUyITP3d4xBm6QZ4jifV-JoeFcxyUrAdTgh-kW548sGm5aT4rZ3CDQuv5_DgFbgDD2sIoeBVftLtx_gVXDHuRG9lIgjwMQlrEsuI-fg-20qgKJ5cN8t6HpNYBNPkPJRYSDgAPpBJu2zk71g8I3n_aq6kB2A5frcPJTBmEjTFrUcKcPzdrxdnLq7zJMgmk1xc7tABP8y8u1-Ik30k8Vm0YtClsNjRjF65aVf8GRoFw_OqkHACs2bOyTnMmtbaXYeFuIgaZxuJO8UvAjYvMMrM-L32iez6OsS-wciShe0JXxatMsxcPYZ8ylONIX3tRZJ0sMCVSBTgKz0ZLvTYDWZy_WPuKXvmPWLNdatNvjGCd-koYnEQJ2hYUUgYEZeybl72YfNG_5jpxcYwx7eeksp1rCVzlyMdqsLINtb0hzD1sWCe1tJ0YZEJMVjySyUSiuDqe2_Q9QxGePDxTmVj8pSaqQ_5gwck_KEd2Zy32eT-VW9EVCpkyWV2pC1yyrFCzNL2AOotxAmRmwk.648-AIpnkhuCfNcUr12aZQ/model.03-1.3574.h5'
model.load_weights('model.h5')

r'''
In this competition, we are identifying and localizing COVID-19 abnormalities on chest radiographs. This is an object detection and classification problem.

For each test image, you will be predicting a bounding box and class for all findings. If you predict that there are no findings, you should create a prediction of "none 1 0 0 1 1" ("none" is the class ID for no finding, and this provides a one-pixel bounding box with a confidence of 1.0).

Further, for each test study, you should make a determination within the following labels:

    'Negative for Pneumonia' 'Typical Appearance' 'Indeterminate Appearance' 'Atypical Appearance'

To make a prediction of one of the above labels, create a prediction string similar to the "none" class above: e.g. atypical 1 0 0 1 1

Please see the Evaluation page for more details about formatting predictions.

The images are in DICOM format, which means they contain additional data that might be useful for visualizing and classifying.
Dataset information

The train dataset comprises 6,334 chest scans in DICOM format, which were de-identified to protect patient privacy. All images were labeled by a panel of experienced radiologists for the presence of opacities as well as overall appearance.

Note that all images are stored in paths with the form study/series/image. The study ID here relates directly to the study-level predictions, and the image ID is the ID used for image-level predictions.

The hidden test dataset is of roughly the same scale as the training dataset.
Files

    train_study_level.csv - the train study-level metadata, with one row for each study, including correct labels.
    train_image_level.csv - the train image-level metadata, with one row for each image, including both correct labels and any bounding boxes in a dictionary format. Some images in both test and train have multiple bounding boxes.
    sample_submission.csv - a sample submission file containing all image- and study-level IDs.

Columns

train_study_level.csv

    id - unique study identifier
    Negative for Pneumonia - 1 if the study is negative for pneumonia, 0 otherwise
    Typical Appearance - 1 if the study has this appearance, 0 otherwise
    Indeterminate Appearance  - 1 if the study has this appearance, 0 otherwise
    Atypical Appearance  - 1 if the study has this appearance, 0 otherwise

train_image_level.csv

    id - unique image identifier
    boxes - bounding boxes in easily-readable dictionary format
    label - the correct prediction label for the provided bounding boxes
ds = pydicom.dcmread('the path to the image')

new_image = ds.pixel_array.astype(float)

As you can see, we do this in two lines of code. The first line with open the Dicom image and the second one will extract the pixel array from it and that’s it for this part. You can see that I added the function “astype” then float. I did that because in the next step we will scale the pixels so if we leave them integers so after the scaling we will have the overflow or the underflow losses. For that, we need to work with float pixels so that we won’t lose the information.
Rescaling the image

Now to use the image we should rescale its pixels and put their values between 0 and 255. So here is the line of code that you need to do that.

scaled_image = (np.maximum(new_image, 0) / new_image.max()) * 255.0

Displaying And Saving The Image

Now the final part which using the image for our needs. The image that we had in the last part was an array so we will use this array to create the image using the Pillow library. But before that, we need to convert it into 8 bits unsigned integer. Here are the commands that you need to create the image from an array:

scaled_image = np.uint8(scaled_image)
final_image = Image.fromarray(scaled_image)

'''

import glob
import pprint
t1 = glob.glob('/kaggle/input/siim-covid19-detection/test/*/*/*.dcm')
pprint.pprint([len(t1), t1[:5]])

def show_dcm_info(study_ids, df):
    '''Show .dcm images along with description.'''
    wandb_logs = []
    
    fig, axes = plt.subplots(nrows=2, ncols=3, figsize=(21,10))

    # Get .dcm paths
    dcm_paths = [glob.glob(f"../input/siim-covid19-detection/train/{study_id}/*/*")[0]
                 for study_id in study_ids]
    datasets = [pydicom.dcmread(path) for path in dcm_paths]
    images = [apply_voi_lut(dataset.pixel_array, dataset) for dataset in datasets]

    # Loop through the information
    for study_id, data, img, i in zip(study_ids, datasets, images, range(2*3)):
        # Fix inverted images
        img = fix_inverted_radiograms(data, img)

        # Below function available in functions section ;)
        label, bbox = get_image_metadata(study_id, df)
        
        # Check for bounding box and add if it's the case
        try: 
            # For no bbox, the list is [nan]
            no_box = math.isnan(bbox[0])
            pass
        except TypeError:
            # Retrieve the bounding box
            all_coords = []
            for box in bbox:
                all_coords.append(return_coords(box))

            for (x1, y1, x2, y2) in all_coords:
                cv2.rectangle(img, (x1, y1), (x2, y2), (0, 80, 255), 15)
                cv2.putText(img, label, (x1, y1-14), 
                            cv2.FONT_HERSHEY_SIMPLEX, 3, (0, 0, 0), 4)
                
        # Plot the image
        x = i // 3
        y = i % 3
        
        axes[x, y].imshow(img, cmap="rainbow")
        axes[x, y].set_title(f"Label: {label} \n Sex: {data.PatientSex} | Body Part: {data.BodyPartExamined}", 
                  fontsize=14, weight='bold')
        axes[x, y].axis('off');
        
        # Save to W&B
        wandb_logs.append(wandb.Image(img, 
                                      caption=f"Label: {label} \n Sex: {data.PatientSex} | Body Part: {data.BodyPartExamined}"))
          
    wandb.log({f"{label}": wandb_logs})
import pydicom
t2 =  pydicom.dcmread(t1[0])
# %% [markdown]
# # Resources 
# - [Mask Spatial Supervision](https://www.kaggle.com/ipythonx/blending-mask-with-x-ray-for-spatial-supervision/notebook?scriptVersionId=66218468): Try to experiemnt by blending with bounding box mask information, (method 1 + method 2).
# - [Add Attention Mechansim](https://www.kaggle.com/ipythonx/tf-keras-ranzcr-multi-attention-efficientnet): Try to add attention function to build hybrid model. 
# - [Out-of-Fold Evaluation](https://www.kaggle.com/ipythonx/optimizing-metrics-out-of-fold-weights-ensemble): Compute the oof each fold and compare the match and non-match prediction in order to emphasize on the weak or minor cases. 