{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Hello everyone, I'm Elad, and welcome to my kernel. In this kernel I will load 1,549 images using a very useful library called VTK.\n# VTK is mostly written in C++ making it incredibly efficient. By using this library you can save loads of memory and time.\n","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd \n\nimport os\nimport sys\n\nimport glob\n\nimport cv2\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf \nimport tensorflow.keras.layers as L\nfrom tensorflow.keras.models import Model\nimport tensorflow.keras.backend as K\n\nfrom joblib import Parallel, delayed\nfrom tqdm import tqdm_notebook\n\n\nimport vtk\nfrom vtk.util import numpy_support\nimport numpy","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"reader = vtk.vtkDICOMImageReader()\ndef get_img(path):\n    reader.SetFileName(path)\n    reader.Update()\n    _extent = reader.GetDataExtent()\n    ConstPixelDims = [_extent[1]-_extent[0]+1, _extent[3]-_extent[2]+1, _extent[5]-_extent[4]+1]\n\n    ConstPixelSpacing = reader.GetPixelSpacing()\n    imageData = reader.GetOutput()\n    pointData = imageData.GetPointData()\n    arrayData = pointData.GetArray(0)\n    ArrayDicom = numpy_support.vtk_to_numpy(arrayData)\n    ArrayDicom = ArrayDicom.reshape(ConstPixelDims, order='F')\n    ArrayDicom = cv2.resize(ArrayDicom,(512,512))\n    return ArrayDicom","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#store unique patient subdirectories\ntr_patient_paths = sorted(glob.glob('/kaggle/input/osic-pulmonary-fibrosis-progression/train/*'))\n#Read and Sort : train csv\ntrain = pd.read_csv('/kaggle/input/osic-pulmonary-fibrosis-progression/train.csv')\ntrain.sort_values(by = 'Patient',inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Retrieve and list paths in sequential order : ie. (dicom1,dicom2 ...)\nall_paths = []\nfor patient_path in tr_patient_paths:\n    organized_paths = glob.glob(patient_path+'/*')\n    organized_paths.sort(key=lambda x: int(x.split('/')[-1].split('.')[0]))\n    all_paths.extend(organized_paths)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Make a path DataFrame\nimage_df = pd.DataFrame(all_paths,columns=['Paths'])\n# Retrieve Patient IDs\nimage_df['Patient'] = image_df['Paths'].apply(lambda x: x.split('/')[5])\n# Retrieve Patient Scan Number : eg. What visit they are getting the scan on\nimage_df['Visit'] = image_df['Paths'].apply(lambda x: int(x.split('/')[-1].split('.')[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Find the time passed\ntrain['min'] = train.groupby('Patient')['Weeks'].transform('min')\ntrain['time_passed'] = (train['Weeks'] - train['min'])+1\n#Find the total treatment period\ntrain['Total_Time_of_Treatment'] = train.groupby('Patient')['time_passed'].transform('max')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"max_visits_tr = dict([*zip(image_df.groupby('Patient')['Visit'].max().index.values.tolist(),image_df.groupby('Patient')['Visit'].max().values.tolist())])\n#Map each patient to their total amount of visits\ntrain['Max_Visits'] = train['Patient'].map(max_visits_tr)\n#Scale each patient visit in the train csv to a particular scan, such as (5 or 136), like in (dicom5 or dicom 136)\ntrain['Visit'] = np.around(train['Max_Visits']*(train['time_passed']/train['Total_Time_of_Treatment']),0).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Drop Patient column so the join works better\nimage_df.drop('Patient',axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Join on the scan from the visit closest to their last measurement --\n#And retrieve the paths corresponding to those scans  \ntr_im_paths = train.join(image_df,lsuffix='Visit', rsuffix='Visit',how='left')['Paths']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Check Shape of Scan Paths\ntr_im_paths.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Display one of the images**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(get_img(tr_im_paths[1500]))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"max_len = len(tr_im_paths)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Here we see that the loading of 1,549 dicom images is within 20 seconds, USING ONLY 1 CORE!!!!","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nX = np.array([Parallel(n_jobs=1)(delayed(get_img)(filename) for filename in tqdm_notebook(tr_im_paths[:max_len]))])[0]\n# X = np.array([Parallel(n_jobs=4)(delayed(get_img)(filename) for filename in tqdm_notebook(tr_im_paths[:max_len]))])[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y  = train['FVC'][:len(X)].astype(np.float32)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Here you can see that the memory taken by all 1,549 images is just 128 BYTES!\n# This is around 145 times less memory than even the target data! \n# The best part of this is that the images are still compatible with Keras and More! ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Shapes are:',X.shape,y.shape)\nprint(\"Memory taken by X and Y in bytes are:\",sys.getsizeof(X),sys.getsizeof(y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"18620/128 ","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}