{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preprocessing and saving images as numpy(.npy) for training models faster \nFor those who're still loading dcm images in the DataLoader\n\n\nSave preprocessed numpy images","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nimport glob\nimport pydicom\nimport numpy as np\nimport cv2\nfrom tqdm import tqdm\nimport math\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2021-08-11T13:22:04.452281Z","iopub.execute_input":"2021-08-11T13:22:04.452634Z","iopub.status.idle":"2021-08-11T13:22:04.456921Z","shell.execute_reply.started":"2021-08-11T13:22:04.452603Z","shell.execute_reply":"2021-08-11T13:22:04.456043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_slices(patient, mri_type='FLAIR'):\n    sorted_imgs = sorted(glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/'+patient+'/'+ mri_type + '/*.dcm'))\n    slices = [pydicom.read_file(s) for s in sorted_imgs]\n    return slices\n\ndef get_pixels_hu(slices):\n    '''\n    Convert pixels to hounsfield units (IMPLEMENTATION SPECIFIC)\n    '''\n    image = np.stack([s.pixel_array for s in slices])\n    image = image.astype(np.int16)\n    image[image == -2000] = 0\n    for slice_number in range(len(slices)):\n        intercept = slices[slice_number].RescaleIntercept\n        slope = slices[slice_number].RescaleSlope\n        if slope != 1:\n            image[slice_number] = slope * image[slice_number].astype(np.float64)\n            image[slice_number] = image[slice_number].astype(np.int16)\n        image[slice_number] += np.int16(intercept)\n    return np.array(image, dtype=np.int16)\n\ndef remove_blanks(hu_images):\n    blanked_images = []\n    for i in range(hu_images.shape[0]):\n        if np.min(hu_images[i]) != np.max(hu_images[i]):\n            blanked_images.append(hu_images[i])\n    return np.array(blanked_images, dtype=np.int16)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T13:22:04.458221Z","iopub.execute_input":"2021-08-11T13:22:04.458673Z","iopub.status.idle":"2021-08-11T13:22:04.475666Z","shell.execute_reply.started":"2021-08-11T13:22:04.458628Z","shell.execute_reply":"2021-08-11T13:22:04.474518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Saving loop","metadata":{}},{"cell_type":"code","source":"# loading labels\nmri_type = 'FLAIR'\nlabels_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv', converters={'BraTS21ID': lambda x: str(x)})\nlabels_df = labels_df.set_index('BraTS21ID')\npatients = os.listdir('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train')\n\nfor patient in tqdm(patients):\n    output = []\n    slices = load_slices(patient)\n    if len(slices) < 10:\n        continue\n    images = get_pixels_hu(slices)\n    images = remove_blanks(images)\n    im = []\n    for j in range(images.shape[0]):\n        im.append(cv2.resize(images[j], (256, 256)))\n    images = np.array(im)\n    # DO PREPROCESSING AUGMENTATION CALLS HERE\n    \n    label = labels_df._get_value(patient, 'MGMT_value')\n    # Change label shape depending on your loss function\n    if label == 1:\n        label = np.array(1)\n    elif label == 0:\n        label = np.array(0)\n        \n    output.append([images, label])\n    all_data_numpy = np.array(output)\n    filename = patient + '.npy'\n    np.save(filename, all_data_numpy)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T13:22:04.47748Z","iopub.execute_input":"2021-08-11T13:22:04.477943Z","iopub.status.idle":"2021-08-11T13:22:13.603257Z","shell.execute_reply.started":"2021-08-11T13:22:04.47791Z","shell.execute_reply":"2021-08-11T13:22:13.60138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Move these to a directory ","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import Dataset\nclass NpData(Dataset):\n    def __init__(self, test=False):\n        self.patients = os.listdir('.') \n        if test == True:\n            self.patients = self.patients[-50:]\n        else:\n            self.patients = self.patients[:-50]\n        \n    def __len__(self):\n        return len(self.patients)\n    \n    def __getitem__(self, x):\n        patient = self.patients[x]\n        # Images in data directory\n        filename = './data/'+ patient\n        image = np.load(filename, allow_pickle=True)\n        \n        images = image[0][0]\n        \n        images = torch.from_numpy(images)\n        \n        images = torch.reshape(images, (1, 64, 256, 256))\n        \n        label = image[0][1].item()\n        label = torch.tensor(abs(label), dtype=torch.float)\n        \n        return images, label","metadata":{"execution":{"iopub.status.busy":"2021-08-11T13:22:13.604226Z","iopub.status.idle":"2021-08-11T13:22:13.604638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}