{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport cv2\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport os\nimport albumentations as A\nimport mlcrate as mlc\nimport shutil\nimport json","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-26T09:49:54.296297Z","iopub.execute_input":"2021-07-26T09:49:54.296999Z","iopub.status.idle":"2021-07-26T09:49:56.448829Z","shell.execute_reply.started":"2021-07-26T09:49:54.296844Z","shell.execute_reply":"2021-07-26T09:49:56.448142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How to make a Dataset with over 20GB\n- https://www.kaggle.com/ksmcg90/miccai-brain-256","metadata":{}},{"cell_type":"markdown","source":"## Add Kaggle Secrets to Notebook","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nusername = user_secrets.get_secret(\"KAGGLE_USERNAME\")\nos.environ['KAGGLE_USERNAME'] = username\nos.environ['KAGGLE_KEY'] = user_secrets.get_secret(\"KAGGLE_KEY\")","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:56.450326Z","iopub.execute_input":"2021-07-26T09:49:56.450869Z","iopub.status.idle":"2021-07-26T09:49:56.924253Z","shell.execute_reply.started":"2021-07-26T09:49:56.450829Z","shell.execute_reply":"2021-07-26T09:49:56.923358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Xray and Resize Functions","metadata":{}},{"cell_type":"code","source":"## Function modified from https://www.kaggle.com/lucamtb/brain-tumor-very-basice-inference\ndef read_xray(path, voi_lut = True, fix_monochrome = True, normalize=False):\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    \n    if normalize:\n        data = data - np.min(data)\n        data = data / np.max(data)\n        #data = (data * 255).astype(np.uint8)\n    else:\n        data = (data / 256).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:56.925891Z","iopub.execute_input":"2021-07-26T09:49:56.92618Z","iopub.status.idle":"2021-07-26T09:49:56.932872Z","shell.execute_reply.started":"2021-07-26T09:49:56.926151Z","shell.execute_reply":"2021-07-26T09:49:56.93198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_function(image_size, original_dir=None, new_dir=None):\n    shape = (image_size, image_size)\n    transform = A.Compose([A.LongestMaxSize(image_size) ,A.PadIfNeeded(*shape, border_mode=0)])\n    def resized(path):\n        image = read_xray(path)\n        image = transform(image=image)['image']\n        new_path = str(path).replace('.dcm', '.npy')\n        if original_dir is not None and new_dir is not None:\n            new_path = new_path.replace(original_dir, new_dir)\n        new_path = Path(new_path)\n        new_path.parent.mkdir(exist_ok=True, parents=True)\n        np.save(new_path, image)\n    return resized","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:56.934471Z","iopub.execute_input":"2021-07-26T09:49:56.93473Z","iopub.status.idle":"2021-07-26T09:49:56.94498Z","shell.execute_reply.started":"2021-07-26T09:49:56.934705Z","shell.execute_reply":"2021-07-26T09:49:56.94417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IM_SIZE = 256\nDATA_DIR = Path('../input/rsna-miccai-brain-tumor-radiogenomic-classification')","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:56.945982Z","iopub.execute_input":"2021-07-26T09:49:56.946249Z","iopub.status.idle":"2021-07-26T09:49:56.954711Z","shell.execute_reply.started":"2021-07-26T09:49:56.946225Z","shell.execute_reply":"2021-07-26T09:49:56.953954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initialize Dataset and add Metadata","metadata":{}},{"cell_type":"code","source":"SAVE_DIR = f'/kaggle/tmp/resized-{IM_SIZE}'\nos.environ['SAVE_DIR'] = SAVE_DIR\nos.makedirs(SAVE_DIR, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:56.95591Z","iopub.execute_input":"2021-07-26T09:49:56.956202Z","iopub.status.idle":"2021-07-26T09:49:56.964864Z","shell.execute_reply.started":"2021-07-26T09:49:56.956174Z","shell.execute_reply":"2021-07-26T09:49:56.964052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets init -p $SAVE_DIR","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:56.966398Z","iopub.execute_input":"2021-07-26T09:49:56.966867Z","iopub.status.idle":"2021-07-26T09:49:58.150256Z","shell.execute_reply.started":"2021-07-26T09:49:56.966826Z","shell.execute_reply":"2021-07-26T09:49:58.149286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(f'{SAVE_DIR}/dataset-metadata.json') as f:\n    data = json.load(f)\n    \ndataset_title = f\"miccai-brain-{IM_SIZE}\"\ndata['title'] = dataset_title\ndata['id'] = f\"{username}/{dataset_title}\"\n\nwith open(f'{SAVE_DIR}/dataset-metadata.json', 'w') as json_file:\n    json.dump(data, json_file)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:58.151659Z","iopub.execute_input":"2021-07-26T09:49:58.151933Z","iopub.status.idle":"2021-07-26T09:49:58.159976Z","shell.execute_reply.started":"2021-07-26T09:49:58.151904Z","shell.execute_reply":"2021-07-26T09:49:58.159355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Copy the csv files in case you want to use the dataset in Google Colab\n- kaggle datasets download -d ksmcg90/miccai-brain-256","metadata":{}},{"cell_type":"code","source":"for path in DATA_DIR.glob('*.csv'):\n    new_path = str(path).replace(str(DATA_DIR), str(SAVE_DIR))\n    shutil.copy(path, new_path)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:58.173214Z","iopub.execute_input":"2021-07-26T09:49:58.173477Z","iopub.status.idle":"2021-07-26T09:49:58.190595Z","shell.execute_reply.started":"2021-07-26T09:49:58.173454Z","shell.execute_reply":"2021-07-26T09:49:58.189577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resizing_func = resize_function(IM_SIZE, str(DATA_DIR), str(SAVE_DIR))","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:58.163125Z","iopub.execute_input":"2021-07-26T09:49:58.163387Z","iopub.status.idle":"2021-07-26T09:49:58.171851Z","shell.execute_reply.started":"2021-07-26T09:49:58.163362Z","shell.execute_reply":"2021-07-26T09:49:58.171072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = list((DATA_DIR).rglob('*.dcm'))","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:49:58.191913Z","iopub.execute_input":"2021-07-26T09:49:58.192315Z","iopub.status.idle":"2021-07-26T09:50:58.354813Z","shell.execute_reply.started":"2021-07-26T09:49:58.192277Z","shell.execute_reply":"2021-07-26T09:50:58.353862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pool = mlc.SuperPool()\nresults = pool.map(resizing_func, paths)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T09:50:58.356211Z","iopub.execute_input":"2021-07-26T09:50:58.356618Z","iopub.status.idle":"2021-07-26T10:15:44.076517Z","shell.execute_reply.started":"2021-07-26T09:50:58.356581Z","shell.execute_reply":"2021-07-26T10:15:44.074897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create or Update Dataset","metadata":{}},{"cell_type":"code","source":"! kaggle datasets create -p $SAVE_DIR -u --dir-mode tar\n#! kaggle datasets version -p $SAVE_DIR -m \"Full Sample\" --dir-mode tar","metadata":{"execution":{"iopub.status.busy":"2021-07-26T10:15:44.080236Z","iopub.execute_input":"2021-07-26T10:15:44.080682Z","iopub.status.idle":"2021-07-26T10:32:26.497533Z","shell.execute_reply.started":"2021-07-26T10:15:44.080641Z","shell.execute_reply":"2021-07-26T10:32:26.496358Z"},"trusted":true},"execution_count":null,"outputs":[]}]}