{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pydicom\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n#import pylibjpeg\nimport pydicom as dicom\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nfrom tensorflow.keras.layers.experimental.preprocessing import Resizing\nimport random\nimport gc \nimport tensorflow as tf\nimport logging\n\n\nfrom threading import Thread\nimport time\nimport os\nfrom pathlib import Path\n\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-24T19:49:06.310714Z","iopub.execute_input":"2021-06-24T19:49:06.311091Z","iopub.status.idle":"2021-06-24T19:49:12.817259Z","shell.execute_reply.started":"2021-06-24T19:49:06.31106Z","shell.execute_reply":"2021-06-24T19:49:12.81618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hi there - I made this to avoid wasting time always loading this big images into memory and scale them down...\n### If you are lazy just go to the Generate Dataset headline and execute the Code below.\n\n### and don't forget to configure this parameters fitting your environment!","metadata":{}},{"cell_type":"code","source":"img_path = 'features/' # WHERE YOU WANT TO STORE THE IMAGES AS .NPY\ntraining_path = \"/kaggle/input/siim-covid19-detection/train/\" #THE FOLDER WITH THE TRAINING DATA\ndata_path = \"/kaggle/input/siim-covid19-detection/\" #KIND OF THE ROOT... WHERE YOUR CSVs ARE\nPath(img_path).mkdir(parents=True, exist_ok=True) # THIS CREATES THE img_path IF THE FOLDER IS NOT THERE YET\n#Path(os.path.join(img_path)).mkdir(parents=True, exist_ok=True)\n\n\nimage_level = pd.read_csv(os.path.join(data_path, \"train_image_level.csv\"))\nstudy_level = pd.read_csv(os.path.join(data_path, \"train_study_level.csv\"))\n\nSIZE=(512, 512) # CHOOSE THE SIZE YOU WANT TO USE\n\nresizer = Resizing(SIZE[0], SIZE[1]) # MAKES IT EASY TO RESIZE... DON'T TOUCH THIS","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:48:28.492738Z","iopub.execute_input":"2021-06-24T19:48:28.493096Z","iopub.status.idle":"2021-06-24T19:48:28.539572Z","shell.execute_reply.started":"2021-06-24T19:48:28.493065Z","shell.execute_reply":"2021-06-24T19:48:28.538393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save as Image and Dataset methods\nThis methods will take the large images and scale them down to SIZE - without loosing the whole pixel information","metadata":{}},{"cell_type":"code","source":"def load_img(study_name):\n    \"\"\"This will return None if the image does not exist\"\"\"\n    npy_path = os.path.join(img_path, (study_name + \".npy\"))\n    file = Path(npy_path)\n    if not file.exists():\n        #print(\"no file\")\n        return None\n    with open(npy_path, 'rb') as f:\n        try:\n            img = np.load(f, allow_pickle=True)\n            return img\n        except:\n            print(\"error loading np file\")\n            return None\n\ndef save_img(img, study_name):\n     with open(os.path.join(img_path, (study_name + \".npy\")), 'wb') as f:\n        np.save(f, resizer(img))\n        \ndef get_img(study, resize=False):\n    \"\"\"\n        This method will read the original file and save it as a npy. \n        if there is allready a npy from a privious call - then it will take the npy directly.\n    \"\"\"\n    img = load_img(study)\n    if img is not None:\n        logging.debug(\"image loaded as npy\")\n        #print(\"loaded\")\n        return img\n    path = os.path.join(training_path, study)\n            #print(path)\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n                    #print(os.path.join(dirname, filename))\n            try:\n                dcm = dicom.dcmread(os.path.join(dirname, filename))\n                img = np.expand_dims(dcm.pixel_array, axis=2).astype(np.float32)\n                img = resizer(img) if resize else img\n                del dcm\n                gc.collect()\n                \n                Thread(target=lambda: save_img(img, study), daemon=False).start()\n                #with open(os.path.join(img_path, (study + \".npy\")), 'wb') as f:\n                    #np.save(f, resizer(img) if resize else img)\n                #cv2.imwrite(os.path.join(img_path, (study + \".png\")), np.multiply(img, 255))\n                logging.debug(\"image saved as npy\")\n               # print(\"saved\")\n                return img\n            except Exception as e:\n                logging.error(e)\n               # print(\"error\")\n                #raise e\n                return None\n            \n            \ndef save_dataset(x, y, x_val, y_val):\n    \"\"\"\n    You can save your Trainingdata directly into big npy files.\n    This saves extra time loading them\n    \"\"\"\n    with open(os.path.join(img_path, (\"x\" + \".npy\")), 'wb') as f:\n        np.save(f, x)\n    with open(os.path.join(img_path, (\"y\" + \".npy\")), 'wb') as f:\n        np.save(f, y)    \n    with open(os.path.join(img_path, (\"x_val\" + \".npy\")), 'wb') as f:\n        np.save(f, x_val)\n    with open(os.path.join(img_path, (\"y_val\" + \".npy\")), 'wb') as f:\n        np.save(f, y_val)    \n        ","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:48:30.666103Z","iopub.execute_input":"2021-06-24T19:48:30.666488Z","iopub.status.idle":"2021-06-24T19:48:30.683527Z","shell.execute_reply.started":"2021-06-24T19:48:30.666453Z","shell.execute_reply":"2021-06-24T19:48:30.682405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generators","metadata":{}},{"cell_type":"code","source":"def scale_data(min_val, max_val, data):\n    new_val = np.divide((np.subtract(data, min_val)),np.subtract(max_val, min_val))\n    return new_val\n\ndef pneumonia_image(only_has = False):\n    if only_has:\n        df = study_level[study_level[\"Negative for Pneumonia\"] == 0].copy()\n    else:\n        df = study_level\n    for index, row in df.iterrows():\n        study = str(row.id).split('_')[0]\n        lable_appearance = row.values[2:].astype(np.float32)\n        lable_has_not_pneumonia = np.asarray([row.values[1]]).astype(np.float32)\n        img = get_img(study)\n        if img is not None:\n            yield img, lable_has_not_pneumonia, lable_appearance\n\nLABLE_HAS_NOT = 0\nLABLE_APPEARANCE = 1\nLABLE_BOTH_CONCATENATED = 2\n                \ndef generate_batch(lable=LABLE_BOTH_CONCATENATED, do_resize=True,  image_width=SIZE[0], image_height=SIZE[1], batch_size=32, only_has=False):\n    \"\"\"\n    This generator can be the direct input for a tensorflow model fit function.\n    depending on the lable you want to predict please chosse one of the following:\n    LABLE_TYPE = 0 only yes/no (has not pneumonia shape (batch_size, 1))\n    LABLE_APPEARANCE = 1 the apperance_type as one_hot_vector shape(batch_size, 3)\n    LABLE_BOT_CONCATENATED = 2 as in the csv shape (batch_size, 4)\n    \n    \"\"\"\n    if lable not in [LABLE_HAS_NOT, LABLE_APPEARANCE, LABLE_BOTH_CONCATENATED]:\n        raise ValueError(\"lable param is not valid!\")\n    imgs = []\n    lables = []\n    counter = 0\n    r = Resizing(image_width, image_height)\n    \n    for img, lable_has_not_pneumonia, lable_appearance in pneumonia_image(only_has):\n        img = scale_data(np.min(img), np.max(img), img.astype(np.float32))\n        imgs.append(r(img) if do_resize else img)\n        if lable == LABLE_HAS_NOT:\n            y_true = lable_has_not_pneumonia\n        elif lable == LABLE_APPEARANCE:\n            y_true = lable_appearance\n        elif lable == LABLE_BOTH_CONCATENATED:\n            y_true = np.concatenate((lable_has_not_pneumonia, lable_appearance), axis=None)\n        lables.append(y_true)\n        counter += 1\n        if counter == batch_size:\n            yield np.array(imgs), np.array(lables)\n            counter = 0\n            del imgs\n            del lables\n            gc.collect()\n            imgs = []\n            lables = []\n\n            \ndef generate_dataset(lable=LABLE_BOTH_CONCATENATED, do_resize=False,  image_width=252, image_height=252, only_has=False, val_percent=0.1, extend_dataset=False):\n    x = []\n    y = []\n    for x_batch, y_batch in generate_batch(lable, do_resize,  image_width, image_height, 1, only_has):\n        x.append(x_batch)\n        y.append(y_batch)\n    s = list(zip(x, y))\n    random.shuffle(s)\n    x, y = zip(*s)\n    x = np.array(x).reshape(len(x), len(x[0][0]), len(x[0][0][0]), 1)\n    y = np.array(y).reshape((-1, len(y[0][0])))\n    set_size = x.shape[0]\n    from_val = set_size - int(set_size * val_percent)\n    x_val = x[from_val:]\n    x = x[:from_val]\n    y_val = y[from_val:]\n    y = y[:from_val]\n    x_max = np.max(x)\n    x_min = np.min(x)\n    #x = scale_data(x_min, x_max, x)\n    #x_val = scale_data(x_min, x_max, x_val)\n\n    \n    if extend_dataset:\n        random_indexes = np.random.choice(x.shape[0], int(x.shape[0] * 0.3))        \n        to_augment_x = x[random_indexes].copy()\n        augmented_x = np.array([random_augmentate_image(i) for i in to_augment_x])\n        augmented_y = y[random_indexes].copy()\n        \n        print(x.shape)\n        print(augmented_x.shape)\n        x = np.concatenate([x, augmented_x],axis=0)\n        y = np.concatenate([y, augmented_y],axis=0)\n        print(x.shape)\n        print(y.shape)\n        \n        s = list(zip(x, y))\n        random.shuffle(s)\n        x, y = zip(*s)\n        x = np.array(x)\n        y = np.array(y)\n    \n    return x, y , x_val, y_val\n\n\ndef random_augmentate_image(image):\n    functions = [zoom_image, shift_image, rotate_image]\n    ret_img = image.copy()\n    for i in np.random.choice(len(functions), len(functions)):\n        if np.random.rand(1) >= 0.5:\n            continue\n        ret_img = functions[i](ret_img)\n    return ret_img\n    \n\ndef zoom_image(img):\n    return tf.keras.preprocessing.image.random_zoom(img, (0.8, 1.2), row_axis=1, col_axis=2, channel_axis=0)\n    \ndef shift_image(img):\n    return tf.keras.preprocessing.image.random_shift(img, 0.15, 0.15, row_axis=1, col_axis=2, channel_axis=0)\n\ndef rotate_image(img):\n    return  tf.keras.preprocessing.image.random_rotation(img, 35, row_axis=1, col_axis=2, channel_axis=0, fill_mode='nearest')\n\ndef flip_image(img):\n    return tf.image.flip_left_right(img)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:48:47.257887Z","iopub.execute_input":"2021-06-24T19:48:47.258296Z","iopub.status.idle":"2021-06-24T19:48:47.28753Z","shell.execute_reply.started":"2021-06-24T19:48:47.258261Z","shell.execute_reply":"2021-06-24T19:48:47.286169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate Dataset\n## THIS WILL TAKE A WHILE - GRAB A COFFEE\n### generate_dataset PARAMS\nlable=LABLE_BOTH_CONCATENATED,&emsp;=>&emsp; Leave this untouched if you want the whole label!<br>\ndo_resize=False,&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;&emsp;=>&emsp; if this is set to False then the image_width and the image_height doesn't matter<br>\nimage_width=252,      &emsp;&emsp;&emsp;&emsp;&emsp;        &emsp;&emsp;&emsp;&emsp;&emsp; &emsp;=>&emsp; for resizing<br>\nimage_height=252,    &emsp;&emsp;&emsp;&emsp;&emsp;       &emsp;&emsp;&emsp;&emsp;&emsp;   &emsp;=>&emsp; for resizing<br>\nonly_has=False,           &emsp;&emsp;&emsp;&emsp;&emsp;     &emsp;=>&emsp; leave this untoched unless you only want features with positive illness<br>\nval_percent=0.1,          &emsp;&emsp;&emsp;&emsp;&emsp;     &emsp;=>&emsp; percentage 0 - 1 for the size of the validation set (will be taken from training)<br>\nextend_dataset=False     &emsp;&emsp;&emsp;&emsp;&emsp;    &emsp;=>&emsp; set to True to augmentate some images (rotation, shift, zoom, flip) they will be added to extend the data","metadata":{}},{"cell_type":"code","source":"x, y, x_val, y_val = generate_dataset(do_resize=False, image_width=768, image_height=768, extend_dataset=False)\nsave_dataset(x=x, y=y, x_val=x_val, y_val=y_val)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:49:18.252454Z","iopub.execute_input":"2021-06-24T19:49:18.252855Z","iopub.status.idle":"2021-06-24T19:49:18.270691Z","shell.execute_reply.started":"2021-06-24T19:49:18.252819Z","shell.execute_reply":"2021-06-24T19:49:18.269727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Dataset\ncall this method to load the saved dataset","metadata":{}},{"cell_type":"code","source":"def load_dataset():\n    with open(os.path.join(img_path, (\"x\" + \".npy\")), 'rb') as f:\n        x = np.load(f, allow_pickle=True)\n    with open(os.path.join(img_path, (\"y\" + \".npy\")), 'rb') as f:\n        y = np.load(f, allow_pickle=True)\n    with open(os.path.join(img_path, (\"x_val\" + \".npy\")), 'rb') as f:\n        x_val = np.load(f, allow_pickle=True)\n    with open(os.path.join(img_path, (\"y_val\" + \".npy\")), 'rb') as f:\n        y_val = np.load(f, allow_pickle=True)\n    return x, y, x_val, y_val\n\nx, y, x_val, y_val = load_dataset()","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:49:22.218421Z","iopub.execute_input":"2021-06-24T19:49:22.218767Z","iopub.status.idle":"2021-06-24T19:49:22.229963Z","shell.execute_reply.started":"2021-06-24T19:49:22.218737Z","shell.execute_reply":"2021-06-24T19:49:22.228948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}