# %% [code] {"execution":{"iopub.status.busy":"2021-12-29T04:29:26.490862Z","iopub.execute_input":"2021-12-29T04:29:26.4915Z"}}
# This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load

import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)

# Input data files are available in the read-only "../input/" directory
# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory

import os
for dirname, _, filenames in os.walk('/kaggle/input'):
    for filename in filenames:
        print(os.path.join(dirname, filename))

# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using "Save & Run All" 
# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session


# %% [code]

import urllib
import zipfile
import os
import sys
import pathlib as Path
from os import listdir, system

os.environ['TF_CPP_MIN_LOG_LEVEL'] = '1'
# Code to filter out verbosity
# 0 = all messages are logged (default behavior)
# 1 = INFO messages are not printed
# 2 = INFO and WARNING messages are not printed
# 3 = INFO, WARNING, and ERROR messages are not printed
# Should be placed before importing tf

# Creating data to view and fit
from sklearn.model_selection import train_test_split
import numpy as np
import pandas as pd
import seaborn as sns
import matplotlib.pyplot as plt
import matplotlib.cm
import matplotlib as mpl
import tensorflow as tf
import tensorflow_hub as hub
import tensorflow_datasets as tfds
import wget
import matplotlib.pyplot as plt
import matplotlib.image as mpimg
import os
import random
import json
from sklearn.metrics import accuracy_score
from helper_functions import make_confusion_matrix
from sklearn.metrics import classification_report
from helper_functions import create_tensorboard_callback, plot_loss_curves, unzip_data, walk_through_dir, compare_historys, calculate_results, wrangle_data_nlp, apply_tokenizer, to_tensor, show_images, f1, precision, recall, check_ds, check_model, windowed_dataset, print_model
from helper_functions import plot_time_series, plot_loss_curves_mae, plot_loss_curves_mse, plot_loss_curves_mape, split_features_labels, plot_loss_curves_no_val, split_features_labels_tfds
from tensorflow.keras.utils import plot_model
tf.random.set_seed(42)
from tensorflow.keras import mixed_precision
mixed_precision.set_global_policy('mixed_float16')  # Set global data policy to mixed float16
# Need tensorflow_io if not already installed
import tensorflow_io as tfio

train_label_path = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'
train_labels = pd.read_csv(train_label_path)
# Transpose to match the form of train data
train_labels['BraTS21ID'] = train_labels['BraTS21ID'].astype(str).apply(lambda x: x.rjust(5, '0'))
train_labels = train_labels.set_index('BraTS21ID')
dict_labels = train_labels.to_dict('dict')
dict_labels = dict_labels['MGMT_value']
train_labels.describe().transpose()

data_path = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/'

for dirpath, dirnames, filenames in os.walk('E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/train'):
    print(f"There are {len(dirnames)} directories and {len(filenames)} images in '{dirpath}'.")
    break

class_names = os.listdir('E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/')

# Read an example image
sample_dir = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/train/00012/T2w/'
sample_list = os.listdir(sample_dir)
image_bytes = tf.io.read_file(sample_dir + random.choice(sample_list))

image = tfio.image.decode_dicom_image(image_bytes, dtype=tf.uint16)

skipped = tfio.image.decode_dicom_image(image_bytes, on_error='skip', dtype=tf.uint8)

lossy_image = tfio.image.decode_dicom_image(image_bytes, scale='auto', on_error='lossy', dtype=tf.uint8)

scaled_image = tfio.image.decode_dicom_image(image_bytes, scale='auto', on_error='lossy', dtype=tf.float32)

fig, axes = plt.subplots(1,3, figsize=(20,10))
axes[0].imshow(np.squeeze(image.numpy()), cmap='gray')
axes[0].set_title('image')
axes[1].imshow(np.squeeze(lossy_image.numpy()), cmap='gray')
axes[1].set_title('lossy image');
axes[2].imshow(np.squeeze(scaled_image.numpy()))
axes[2].set_title('scaled image');

# Try saving 1 file
test_file = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/test/00037/T1w/Image-10.dcm'
test_bytes = tf.io.read_file(test_file)
scaled_image = tfio.image.decode_dicom_image(test_bytes, scale='auto', on_error='lossy', dtype=tf.float32)
scaled_image = tf.expand_dims(tf.squeeze(scaled_image), axis=2)
save_path = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/'
tf.keras.utils.save_img((save_path + 'trial_image.jpeg'), scaled_image, file_format='jpeg')

# Try df image
try_path = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/test_jpg/trial_image.jpeg'
try_df = pd.DataFrame({'path': try_path, 'MGMT_value': 'YES' }, index=[1])
from tensorflow.keras.preprocessing.image import ImageDataGenerator
datagen= ImageDataGenerator(rescale=1./255)
train_data_augmented = datagen.flow_from_dataframe(try_df, directory=None, x_col = 'path', y_col = 'MGMT_value',
                                               batch_size=32,
                                               target_size=(500, 500),
                                               class_mode="categorical",
                                               shuffle=True,  # good practice to do
                                               seed=42)  # Takes the path to a directory & generates batches of
                                                            # augmented data.
# It works. In conclusion, ImageDataGenerator doesn't recognize dcim image => need to convert everything into jpeg and dump them somewhere

def to_jpeg(image_path, image_name, save_path):
    """
    Convert dcim to jpeg as float32
    image = absolute/relative path to image
    """
    import tensorflow_io as tfio
    image_bytes = tf.io.read_file(image_path)
    scaled_image = tfio.image.decode_dicom_image(image_bytes, scale='auto', on_error='lossy', dtype=tf.float32)
    scaled_image = tf.expand_dims(tf.squeeze(scaled_image), axis=2)
    tf.keras.utils.save_img((save_path + image_name + '.jpeg'), scaled_image, file_format='jpeg')
    return None

# For this problem, you need to use flow_from_dataframe
# DataFrame format: x_col = filenames or absolute path if directory is None
#                   y_col = string or list that has the label
train_dir = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/train'
all_files = []
list_of_subject = os.listdir(train_dir)
list_of_subset = os.listdir(train_dir + '/' + list_of_subject[0] + '/')
subset_dir = train_dir + '/' + list_of_subject[0] + '/' + list_of_subset[0] + '/'
for i in list_of_subject:
    for j in list_of_subset:
        subject_dict = {}
        dirpath = train_dir + '/' + i + '/' + j + '/'
        list_of_file_name = [os.path.join(dirpath, file) for file in os.listdir(dirpath)]
        subject_dict['BraTS21ID'] = i
        subject_dict['type'] = j
        # Strip extension from filename
        subject_dict['filename'] = [file.split('.')[0] for file in os.listdir(dirpath)]
        subject_dict['path'] = list_of_file_name
        all_files.append(subject_dict)
all_files = pd.DataFrame(all_files)
# Assigning labels to each patient (BraTS21ID)
all_files['MGMT_value'] = all_files['BraTS21ID'].copy()
all_files['MGMT_value'] = all_files['MGMT_value'].map(dict_labels)

# Get all links and label in one df
all_links = pd.DataFrame(columns=all_files.columns[2:])
# Get df of image link and label
for i in range(len(all_files)):
    image_link = {}
    image_link['path'] = list(all_files['path'][i])
    joined_name = all_files['BraTS21ID'][i] + '_' + all_files['type'][i] + '_'
    filename_list = list(all_files['filename'][i])
    joined_filename = [joined_name + x for x in filename_list]
    image_link['filename'] = joined_filename
    temp_container = np.empty(len(all_files['path'][i]))
    temp_container.fill(all_files['MGMT_value'][i])
    image_link['MGMT_value'] = list(temp_container)
    all_links = pd.concat([all_links, pd.DataFrame(image_link)], axis=0)
all_links.count()

dcim_path = all_links.copy()
dcim_path.pop('MGMT_value')
# Generating jpeg
save_path = 'E:/pythonProject2/Tensorflow Dataset/rsna-miccai-brain-tumor-radiogenomic-classification/train_jpeg/'
for i in range(len(all_links)):
    to_jpeg(dcim_path['path'].iloc[i], dcim_path['filename'].iloc[i], save_path)



binary = all_links['MGMT_value']
true_values = binary == 1.0  # if wv == -9999.0, it's True, otherwise False
binary[true_values] = 'YES'  # Replace wv values where it's True. This method is ingeneous!

binary = all_links['MGMT_value']
true_values = binary == 0.0  # if wv == -9999.0, it's True, otherwise False
binary[true_values] = 'NO'  # Replace wv values where it's True. This method is ingeneous!

# Read in dcim images
def create_dataset_tf(img_folder):
    class_name = []
    tf_img_data_array = []

    for dir1 in os.listdir(img_folder):
        for file in os.listdir(os.path.join(img_folder, dir1)):
            image = os.path.join(img_folder, dir1, file)
            image = tf.io.read_file(image)
            image = tf.io.decode_jpeg(image, channels=3)
            image = tf.image.resize(image, (200, 200))
            image = tf.cast(image / 255., tf.float32)
            tf_img_data_array.append(image)
            class_name.append(dir1)
    return tf.stack(tf_img_data_array, axis=0), class_nameimg_folder=r'CV/Intel_Images/seg_train/seg_train'
    tf_img_data, class_name = create_dataset_tf(img_folder)

# %% [markdown]
# Not ready for run yet. This is a note from local Python file

# %% [markdown]
# 

# %% [markdown]
# 