{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\n\nfrom numpy import random\n\nimport pandas as pd \n\nimport os\n\nimport matplotlib.pyplot as plt\n\nimport pydicom\n\nfrom glob import glob\n\nimport gc\n\nimport plotly.express as px\n\nimport cv2  \n\nimport sys","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-14T08:06:43.645072Z","iopub.execute_input":"2021-09-14T08:06:43.645822Z","iopub.status.idle":"2021-09-14T08:06:43.651982Z","shell.execute_reply.started":"2021-09-14T08:06:43.645774Z","shell.execute_reply":"2021-09-14T08:06:43.651058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'","metadata":{"execution":{"iopub.status.busy":"2021-09-14T08:06:43.653582Z","iopub.execute_input":"2021-09-14T08:06:43.653839Z","iopub.status.idle":"2021-09-14T08:06:43.665693Z","shell.execute_reply.started":"2021-09-14T08:06:43.653808Z","shell.execute_reply":"2021-09-14T08:06:43.664783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def z_score(im):\n\n    mask_im = im>im.mean()\n    logical_mask = mask_im>0.\n    mean = im[logical_mask].mean()\n    std = im[logical_mask].std()\n    return (im-mean)/std\n\ndef dcm_to_array(cohort, case, mpMRI):\n    \n    p = f'{PATH}/{cohort}/{case}/{mpMRI}/*.dcm'\n    sorted_files = sorted(glob(p),key=lambda f: int(f.split('Image-')[1].split('.')[0]))\n    slices = [pydicom.dcmread(f) for f in sorted_files]\n    images = np.array([i.pixel_array for i in slices])\n    del slices, sorted_files, p\n    gc.collect()\n    \n    return images\n\ndef remove_black_vox(images):\n    \n    min=np.array(np.nonzero(images)).min(axis=1)\n    max=np.array(np.nonzero(images)).max(axis=1)\n    images_ = images[min[0]:max[0],min[1]:max[1],min[2]:max[2]]\n    \n    return images_\n    \ndef resize_(arr):\n    rsz = []\n    \n    for i in arr:\n        k = cv2.resize(i, (128, 128), interpolation = cv2.INTER_AREA)\n        rsz.append(k)\n        \n    return np.array(rsz)\n    \ndef extract_slices(img_arr, rt=32):\n    \n    _, h, w = img_arr.shape\n    \n    sliced =[]\n    \n    if len(img_arr)>rt:\n        \n        indices = np.unique(np.linspace(0, len(img_arr)-1, rt).astype(int))\n        for i in range(len(indices)-1):\n            sliced.append(np.mean(img_arr[indices[i]:indices[i+1]], axis=0))\n        del indices\n        \n    else:\n        \n        sliced = np.copy(img_arr)\n        \n    if len(sliced)<rt:\n        \n        img_arr = np.copy(sliced)\n        \n        if (rt - len(img_arr))%2==1:\n            a = np.zeros([(rt - len(img_arr))//2, h, w])\n            b = np.zeros([((rt - len(img_arr))//2)+1, h, w])\n            sliced = np.concatenate((a,img_arr,b), axis = 0)\n            del a,b\n            \n            \n        else:\n            a = np.zeros([(rt - len(img_arr))//2, h, w])\n            sliced = np.concatenate((a,img_arr,a), axis = 0)\n            del a\n            gc.collect()\n            \n    return np.array(sliced)\n\n\ndef pp(cohort, case, mpmri):\n    vet = dcm_to_array(cohort, case, mpmri)\n    \n    if np.sum(vet[int(len(vet)//2)])>1000:\n        \n        ada = np.copy(vet)\n        resized = z_score(resize_(extract_slices(remove_black_vox(ada))))\n        \n        resized = np.reshape(resized, (1, 32, 128, 128))\n    \n        del ada\n        gc.collect()\n        \n    else:\n        resized = np.zeros([1, 32, 128, 128])\n    \n    return resized","metadata":{"execution":{"iopub.status.busy":"2021-09-14T08:38:11.359857Z","iopub.execute_input":"2021-09-14T08:38:11.361030Z","iopub.status.idle":"2021-09-14T08:38:11.384283Z","shell.execute_reply.started":"2021-09-14T08:38:11.360983Z","shell.execute_reply":"2021-09-14T08:38:11.383611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/data-label/df.csv')","metadata":{"execution":{"iopub.status.busy":"2021-09-14T08:38:13.750536Z","iopub.execute_input":"2021-09-14T08:38:13.751217Z","iopub.status.idle":"2021-09-14T08:38:13.758920Z","shell.execute_reply.started":"2021-09-14T08:38:13.751180Z","shell.execute_reply":"2021-09-14T08:38:13.758177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def PreProcessing(cohort, cases, train_lab):\n    \n    np_data = []\n    np_data_labels = []\n    mpMRI = ['FLAIR', 'T1w', 'T1wCE', 'T2w']\n    \n    for i in range(len(cases)):\n        a = pp(cohort, cases[i], mpMRI[0])\n        b = pp(cohort, cases[i], mpMRI[1])\n        c = pp(cohort, cases[i], mpMRI[2])\n        d = pp(cohort, cases[i], mpMRI[3])\n        e = np.concatenate((a, b, c, d), axis=0)\n        np_data.append(e)\n        np_data_labels.append([train_lab[i]])\n        del a  \n        gc.collect()\n    \n    return np.array(np_data), np.array(np_data_labels)\n\ncases = ['{0}'.format(str(i).zfill(5)) for i in df['BraTS21ID']]\nlabels = list(df['MGMT_value'])\n\nfor i in range(0, 500, 100):\n    data_, labels_ = PreProcessing('train', cases[i:i+100], labels[i:i+100])\n    name = 'flair_data_train'+str(i)+'_'+str(i+100)\n    np.savez(name, data_, labels_)\n    del data_, labels_\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T08:51:12.012939Z","iopub.execute_input":"2021-09-14T08:51:12.013289Z","iopub.status.idle":"2021-09-14T10:04:30.053904Z","shell.execute_reply.started":"2021-09-14T08:51:12.013259Z","shell.execute_reply":"2021-09-14T10:04:30.052437Z"},"trusted":true},"execution_count":null,"outputs":[]}]}