{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nimport numpy as np\nimport sys\nnp.set_printoptions(threshold=sys.maxsize)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:06.460655Z","iopub.execute_input":"2021-09-12T19:53:06.461217Z","iopub.status.idle":"2021-09-12T19:53:12.044346Z","shell.execute_reply.started":"2021-09-12T19:53:06.461168Z","shell.execute_reply":"2021-09-12T19:53:12.043544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport random\ndf = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv', header=0, names=['id','value'], dtype=object)\ndf = df[~df.id.isin([\"00109\", \"00123\", \"00709\"])]","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:12.046248Z","iopub.execute_input":"2021-09-12T19:53:12.046485Z","iopub.status.idle":"2021-09-12T19:53:12.067511Z","shell.execute_reply.started":"2021-09-12T19:53:12.046454Z","shell.execute_reply":"2021-09-12T19:53:12.066833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:12.068733Z","iopub.execute_input":"2021-09-12T19:53:12.068998Z","iopub.status.idle":"2021-09-12T19:53:12.086049Z","shell.execute_reply.started":"2021-09-12T19:53:12.068967Z","shell.execute_reply":"2021-09-12T19:53:12.085456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://stackoverflow.com/a/4836734/8245487\ndef natural_sort(l): \n    convert = lambda text: int(text) if text.isdigit() else text.lower()\n    alphanum_key = lambda key: [convert(c) for c in re.split('([0-9]+)', key)]\n    return sorted(l, key=alphanum_key)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:12.088652Z","iopub.execute_input":"2021-09-12T19:53:12.088841Z","iopub.status.idle":"2021-09-12T19:53:12.093308Z","shell.execute_reply.started":"2021-09-12T19:53:12.088819Z","shell.execute_reply":"2021-09-12T19:53:12.092603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pydicom\nimport pandas as pd\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm import tqdm\nimport binascii\nfrom PIL import Image\n\nINPUT = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\n\nif not os.path.exists('./train'):\n    os.makedirs('./train')\n    \n\nif not os.path.exists('./test'):\n    os.makedirs('./test')\n\n\ndef get_dicom_files(input_dir, dataset='train'):\n    for subdir, dirs, files in os.walk(f\"{input_dir}/{dataset}\"):\n        if len(files) == 0:\n            continue\n        filename = natural_sort(files)[len(files)//2] #take middle most image -- FLAIR DCM file per training item.\n        filepath = os.path.join(subdir, filename)\n        \n        if filepath.endswith(\".dcm\") and \"FLAIR\" in filepath:\n            cur_id = subdir.split('/')[-2]\n            outpath = os.path.join(f'./{dataset}',f'{cur_id}.png')\n            \n            process_dicom(filepath, outpath)\n\ndef process_dicom(path, outpath):\n    dicom = pydicom.read_file(path)\n    data = apply_voi_lut(dicom.pixel_array, dicom)\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    \n    height = len(data)\n    width = len(data[0])\n    \n    pixels_out = []\n    for row in data:\n        pixels_out.extend(row)\n    assert(len(pixels_out) == height * width)\n    \n    image_out = Image.new('L', (width, height))\n    image_out.putdata(pixels_out)\n    image_out.save(outpath)\n\nget_dicom_files(INPUT, 'train')\nget_dicom_files(INPUT, 'test')\n\n\n#     final = pd.DataFrame(final)\n#     final.to_csv(f\"{args['output']}/dicom_meta_{args['dataset']}.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:27.541409Z","iopub.execute_input":"2021-09-12T19:53:27.542199Z","iopub.status.idle":"2021-09-12T19:55:00.747066Z","shell.execute_reply.started":"2021-09-12T19:53:27.542163Z","shell.execute_reply":"2021-09-12T19:55:00.746331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for id_num in df.id:\n    full_path = './train/{}.png'.format(id_num)\n    df.loc[df.id == id_num, 'file'] = full_path\n    ","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:55:00.748656Z","iopub.execute_input":"2021-09-12T19:55:00.748910Z","iopub.status.idle":"2021-09-12T19:55:01.072709Z","shell.execute_reply.started":"2021-09-12T19:55:00.748871Z","shell.execute_reply":"2021-09-12T19:55:01.072001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:12.504027Z","iopub.execute_input":"2021-09-12T19:53:12.504300Z","iopub.status.idle":"2021-09-12T19:53:12.524263Z","shell.execute_reply.started":"2021-09-12T19:53:12.504269Z","shell.execute_reply":"2021-09-12T19:53:12.523397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = ImageDataLoaders.from_df(df, item_tfms=Resize(224), bs=64, label_col =1, fn_col=2, path='')","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:55:01.074027Z","iopub.execute_input":"2021-09-12T19:55:01.074301Z","iopub.status.idle":"2021-09-12T19:55:01.145554Z","shell.execute_reply.started":"2021-09-12T19:55:01.074267Z","shell.execute_reply":"2021-09-12T19:55:01.144816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:53:20.215763Z","iopub.execute_input":"2021-09-12T19:53:20.216045Z","iopub.status.idle":"2021-09-12T19:53:20.237008Z","shell.execute_reply.started":"2021-09-12T19:53:20.216017Z","shell.execute_reply":"2021-09-12T19:53:20.235147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch \nimport torch.nn as nn\nimport torch.nn.functional as F\n\n\nclass Net(nn.Module):\n    def __init__(self, pretrained=False):\n        super().__init__()\n        self.conv1 = nn.Conv2d(3, 6, 5)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(6, 16, 5)\n        self.fc1 = nn.Linear(16 * 5 * 5, 120)\n        self.fc2 = nn.Linear(120, 84)\n        self.fc3 = nn.Linear(84, 10)\n\n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.pool(F.relu(self.conv2(x)))\n        x = torch.flatten(x, 1) # flatten all dimensions except batch\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        return x","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:55:13.467043Z","iopub.execute_input":"2021-09-12T19:55:13.467748Z","iopub.status.idle":"2021-09-12T19:55:13.475817Z","shell.execute_reply.started":"2021-09-12T19:55:13.467712Z","shell.execute_reply":"2021-09-12T19:55:13.475169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Net())","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:55:13.740560Z","iopub.execute_input":"2021-09-12T19:55:13.741127Z","iopub.status.idle":"2021-09-12T19:55:13.747778Z","shell.execute_reply.started":"2021-09-12T19:55:13.741075Z","shell.execute_reply":"2021-09-12T19:55:13.746883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nlearn = cnn_learner(dls, Net, metrics=[error_rate, accuracy], model_dir=\"/tmp/model/\").to_fp16()","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:55:14.874829Z","iopub.execute_input":"2021-09-12T19:55:14.875091Z","iopub.status.idle":"2021-09-12T19:55:14.888555Z","shell.execute_reply.started":"2021-09-12T19:55:14.875063Z","shell.execute_reply":"2021-09-12T19:55:14.887875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.loss_func","metadata":{"execution":{"iopub.status.busy":"2021-09-12T19:55:16.754272Z","iopub.execute_input":"2021-09-12T19:55:16.754830Z","iopub.status.idle":"2021-09-12T19:55:16.760227Z","shell.execute_reply.started":"2021-09-12T19:55:16.754792Z","shell.execute_reply":"2021-09-12T19:55:16.759353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:39:56.59021Z","iopub.execute_input":"2021-08-12T00:39:56.590548Z","iopub.status.idle":"2021-08-12T00:40:22.735268Z","shell.execute_reply.started":"2021-08-12T00:39:56.590512Z","shell.execute_reply":"2021-08-12T00:40:22.734433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(5, lr_max=1e-2)","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:22.736597Z","iopub.execute_input":"2021-08-12T00:40:22.73696Z","iopub.status.idle":"2021-08-12T00:40:47.339635Z","shell.execute_reply.started":"2021-08-12T00:40:22.736914Z","shell.execute_reply":"2021-08-12T00:40:47.338833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.show_results()","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:47.341073Z","iopub.execute_input":"2021-08-12T00:40:47.341432Z","iopub.status.idle":"2021-08-12T00:40:48.267535Z","shell.execute_reply.started":"2021-08-12T00:40:47.341391Z","shell.execute_reply":"2021-08-12T00:40:48.266514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.DataFrame(columns=['id', 'value'])\ndf_test.id = os.listdir(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/\")","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:48.269094Z","iopub.execute_input":"2021-08-12T00:40:48.269444Z","iopub.status.idle":"2021-08-12T00:40:48.280308Z","shell.execute_reply.started":"2021-08-12T00:40:48.269404Z","shell.execute_reply":"2021-08-12T00:40:48.279421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for id_num in df_test.id:\n    full_path = './test/{}.png'.format(id_num)\n    prediction = learn.predict(full_path)\n    print(prediction)\n    probability = prediction[2][1].item()\n    print(probability)\n    df_test.loc[df_test.id==id_num, 'value'] = probability","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:48.281803Z","iopub.execute_input":"2021-08-12T00:40:48.282438Z","iopub.status.idle":"2021-08-12T00:40:50.553702Z","shell.execute_reply.started":"2021-08-12T00:40:48.282389Z","shell.execute_reply":"2021-08-12T00:40:50.552945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:50.5573Z","iopub.execute_input":"2021-08-12T00:40:50.559288Z","iopub.status.idle":"2021-08-12T00:40:50.573952Z","shell.execute_reply.started":"2021-08-12T00:40:50.559244Z","shell.execute_reply":"2021-08-12T00:40:50.572991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.value.min()","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:50.577478Z","iopub.execute_input":"2021-08-12T00:40:50.579398Z","iopub.status.idle":"2021-08-12T00:40:50.588698Z","shell.execute_reply.started":"2021-08-12T00:40:50.579358Z","shell.execute_reply":"2021-08-12T00:40:50.587785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.value.max()","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:50.590203Z","iopub.execute_input":"2021-08-12T00:40:50.590882Z","iopub.status.idle":"2021-08-12T00:40:50.596642Z","shell.execute_reply.started":"2021-08-12T00:40:50.590845Z","shell.execute_reply":"2021-08-12T00:40:50.595773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.rename(columns={'id':'BraTS21ID','value':'MGMT_value'}).to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-12T00:40:50.598125Z","iopub.execute_input":"2021-08-12T00:40:50.598836Z","iopub.status.idle":"2021-08-12T00:40:50.611428Z","shell.execute_reply.started":"2021-08-12T00:40:50.598801Z","shell.execute_reply":"2021-08-12T00:40:50.610486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}