{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# this version has labels\n# !pip install --quiet torchio\n\n# this version has no labels\n!pip install git+https://github.com/laynr/torchio.git@plot_animation","metadata":{"execution":{"iopub.status.busy":"2021-10-08T14:34:23.197124Z","iopub.execute_input":"2021-10-08T14:34:23.198028Z","iopub.status.idle":"2021-10-08T14:34:38.229993Z","shell.execute_reply.started":"2021-10-08T14:34:23.197904Z","shell.execute_reply":"2021-10-08T14:34:38.229408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport torchio as tio\nfrom pathlib import Path\nimport multiprocessing as mp\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom torch.utils.data import random_split, DataLoader\n\nplt.rcParams[\"figure.figsize\"] = (12, 10)\n\nout_dir      = Path.cwd() / \"dataset\"\ndata_dir     = Path('/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification')\ntraining_dir = data_dir / 'train'","metadata":{"execution":{"iopub.status.busy":"2021-10-08T14:34:38.231780Z","iopub.execute_input":"2021-10-08T14:34:38.232023Z","iopub.status.idle":"2021-10-08T14:34:40.515099Z","shell.execute_reply.started":"2021-10-08T14:34:38.231998Z","shell.execute_reply":"2021-10-08T14:34:40.513912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get patients\ndef get_patients(patients_dir, demo=True):\n    dir_list = training_dir.glob('*')\n    patients = [x.name for x in dir_list if x.is_dir()]\n    \n    if demo:\n        patients = patients[:5]\n\n    # Remove cases the competion host said to exclude \n    # https://www.kaggle.com/c/rsna-miccai-brain-tumor-radiogenomic-classification/discussion/262046\n    if '00109' in patients: patients.remove('00109')\n    if '00123' in patients: patients.remove('00123')\n    if '00709' in patients: patients.remove('00709')\n        \n    return patients\n\npatients = get_patients(training_dir, demo=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T14:34:40.516601Z","iopub.execute_input":"2021-10-08T14:34:40.517139Z","iopub.status.idle":"2021-10-08T14:34:40.563509Z","shell.execute_reply.started":"2021-10-08T14:34:40.517112Z","shell.execute_reply":"2021-10-08T14:34:40.562806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_preparation(patients):\n    subjects  = []\n    labels_df = pd.read_csv(data_dir / 'train_labels.csv', index_col=0)\n    # loop thru patients\n    for patient in patients:\n        # get label for patient\n        label = labels_df._get_value(int(patient), 'MGMT_value')\n        # create subject object for each patient\n        subject = tio.Subject(\n            BraTS21ID=patient,\n            MGMT_value=label,\n            FLAIR=tio.ScalarImage(training_dir / patient / 'FLAIR',),\n            T1w=tio.ScalarImage(training_dir / patient / 'T1w',),\n            T1wCE=tio.ScalarImage(training_dir / patient / 'T1wCE',),\n            T2w=tio.ScalarImage(training_dir / patient / 'T2w',),\n         )\n        # add subject object to subjects list\n        subjects.append(subject)\n\n    # preprocessing transforms\n    preprocessing_transforms = tio.Compose([\n        tio.ToCanonical(),\n        tio.Resample(1, image_interpolation='bspline'),\n        tio.Resample('T1w', image_interpolation='nearest'),\n        #tio.RescaleIntensity((-1, 1)),\n        tio.CropOrPad((280, 280, 264)),\n        #tio.CropOrPad((128, 128, 64))\n        #tio.OneHot(),\n    ])\n        \n\n    # create datasets from transformed subjects\n    dataset = tio.SubjectsDataset(subjects, transform=preprocessing_transforms)\n    print(f'patients :{len(dataset)}')\n\n    \n    return dataset\n\n# create training and validation datasets    \ndataset = data_preparation(patients) ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-08T14:34:40.564920Z","iopub.execute_input":"2021-10-08T14:34:40.565117Z","iopub.status.idle":"2021-10-08T14:34:40.623335Z","shell.execute_reply.started":"2021-10-08T14:34:40.565092Z","shell.execute_reply":"2021-10-08T14:34:40.622388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create png\ndef preprocess_dataset(dataset, out_dir, parallel=True):\n    if parallel:\n        loader = DataLoader(\n            dataset,\n            num_workers=mp.cpu_count(),\n            collate_fn=lambda x: x[0],\n        )\n        iterable = loader\n    else:\n        iterable = dataset\n    for subject in tqdm(iterable):\n        if 0 == subject[\"MGMT_value\"]:\n            class_dir = out_dir / '0'\n        if 1 == subject[\"MGMT_value\"]:\n            class_dir = out_dir / '1'\n            \n        class_dir.mkdir(parents=True, exist_ok=True)\n        filename = class_dir / f'{subject[\"BraTS21ID\"]}.png'\n        subject.plot(reorient=False, output_path=filename, show=False)\n\n# save as png\npreprocess_dataset(dataset, out_dir)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-08T14:34:40.625416Z","iopub.execute_input":"2021-10-08T14:34:40.626560Z","iopub.status.idle":"2021-10-08T14:37:22.949904Z","shell.execute_reply.started":"2021-10-08T14:34:40.626501Z","shell.execute_reply":"2021-10-08T14:37:22.948807Z"},"trusted":true},"execution_count":null,"outputs":[]}]}