{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2 \nimport copy\nimport math\nimport sys\nimport gc\nimport numpy as np\nimport pandas as pd\n\nimport os\nfrom os import listdir\n\nfrom matplotlib import pyplot as plt","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2021-10-15T14:43:33.313052Z","iopub.execute_input":"2021-10-15T14:43:33.313508Z","iopub.status.idle":"2021-10-15T14:43:33.489027Z","shell.execute_reply.started":"2021-10-15T14:43:33.313385Z","shell.execute_reply":"2021-10-15T14:43:33.488061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_files_number(x, y, z):\n    path = x; file = y; file_1 = z\n    c = 0\n    for a in os.listdir(path + \"/\" + file + \"/\" + file_1):\n        c = c + 1\n    return(c)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:43:33.490668Z","iopub.execute_input":"2021-10-15T14:43:33.490898Z","iopub.status.idle":"2021-10-15T14:43:33.496875Z","shell.execute_reply.started":"2021-10-15T14:43:33.490871Z","shell.execute_reply":"2021-10-15T14:43:33.495744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npath = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train'\nnumber_of_files_Same = []\ncount_file_in_same_subfolder = []\nfor file in os.listdir(path):                                \n    temp = []\n    for file_1 in os.listdir(path + \"/\" + file): \n        count_file_in_each_folder = count_files_number(path, file, file_1)\n        temp.append(count_file_in_each_folder)\n    if temp[0]==temp[1] and temp[0]==temp[2] and temp[0]==temp[3]:\n        number_of_files_Same.append(file)\n        count_file_in_same_subfolder.append(temp)\ncount_file_in_same_subfolder = np.array(count_file_in_same_subfolder)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:43:33.498346Z","iopub.execute_input":"2021-10-15T14:43:33.498597Z","iopub.status.idle":"2021-10-15T14:44:29.653274Z","shell.execute_reply.started":"2021-10-15T14:43:33.498570Z","shell.execute_reply":"2021-10-15T14:44:29.652255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del temp, file, file_1, count_files_number\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:29.655752Z","iopub.execute_input":"2021-10-15T14:44:29.656544Z","iopub.status.idle":"2021-10-15T14:44:29.759756Z","shell.execute_reply.started":"2021-10-15T14:44:29.656495Z","shell.execute_reply":"2021-10-15T14:44:29.758816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv_dataframe = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\nlabel_dataframe = copy.deepcopy(csv_dataframe)\n\nlabel_dataframe.insert(1, 'With_zero', np.ones(len(label_dataframe['BraTS21ID']), dtype=str))\n\ncount = 0\nfor i in label_dataframe['BraTS21ID']:\n    if i>=0 and i<10:                \n        i = '0000' + str(i)\n    elif i>=10 and i<100:            \n        i = '000' + str(i)\n    elif i>=100 and i<1000:          \n        i = '00' + str(i)\n    elif i>=1000 and i<10000:        \n        i = '0' + str(i)\n    label_dataframe['With_zero'][count] = i\n    count = count + 1","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:29.761427Z","iopub.execute_input":"2021-10-15T14:44:29.761744Z","iopub.status.idle":"2021-10-15T14:44:29.952363Z","shell.execute_reply.started":"2021-10-15T14:44:29.761705Z","shell.execute_reply":"2021-10-15T14:44:29.950544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del csv_dataframe, count, label_dataframe['BraTS21ID']\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:29.953414Z","iopub.execute_input":"2021-10-15T14:44:29.953613Z","iopub.status.idle":"2021-10-15T14:44:30.054579Z","shell.execute_reply.started":"2021-10-15T14:44:29.953588Z","shell.execute_reply":"2021-10-15T14:44:30.053633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv_dataframe = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\nlist_label = []\nfor i in range(len(label_dataframe['With_zero'])):\n    for j in range(len(number_of_files_Same)):\n        if label_dataframe['With_zero'][i] == number_of_files_Same[j]:\n            list_label.append(label_dataframe.iloc[i]['MGMT_value']) \nlist_label = np.array(list_label)\ns = 0; t = 0\nfor i in list_label:\n    if i == 1:\n        s = s + 1\n    else:\n        t = t + 1","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:30.056079Z","iopub.execute_input":"2021-10-15T14:44:30.056321Z","iopub.status.idle":"2021-10-15T14:44:30.355782Z","shell.execute_reply.started":"2021-10-15T14:44:30.056283Z","shell.execute_reply":"2021-10-15T14:44:30.355148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del csv_dataframe, s, t, label_dataframe\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:30.356914Z","iopub.execute_input":"2021-10-15T14:44:30.357841Z","iopub.status.idle":"2021-10-15T14:44:30.449244Z","shell.execute_reply.started":"2021-10-15T14:44:30.357787Z","shell.execute_reply":"2021-10-15T14:44:30.448365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_list_label = []\nfor i in range(len(list_label)):\n    if list_label[i] == 1:\n        a = np.ones(count_file_in_same_subfolder[i][0], dtype=int)\n        new_list_label.append(a)             \n        new_list_label.append(a)\n        new_list_label.append(a)\n        new_list_label.append(a)\n    elif list_label[i] == 0:\n        b = np.zeros(count_file_in_same_subfolder[i][0], dtype=int)\n        new_list_label.append(b)             \n        new_list_label.append(b)\n        new_list_label.append(b)\n        new_list_label.append(b)\n        \nnew_list_label = np.array(new_list_label)\n\nnew_list_label_2 = []                 \nfor i in new_list_label:\n    for j in i:\n        new_list_label_2.append(j)\nnew_list_label_2 = np.array(new_list_label_2)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:30.450527Z","iopub.execute_input":"2021-10-15T14:44:30.451278Z","iopub.status.idle":"2021-10-15T14:44:30.466373Z","shell.execute_reply.started":"2021-10-15T14:44:30.451233Z","shell.execute_reply":"2021-10-15T14:44:30.465352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del list_label, count_file_in_same_subfolder, a, b, i, j\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:30.469102Z","iopub.execute_input":"2021-10-15T14:44:30.469342Z","iopub.status.idle":"2021-10-15T14:44:30.569169Z","shell.execute_reply.started":"2021-10-15T14:44:30.469315Z","shell.execute_reply":"2021-10-15T14:44:30.568085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom pydicom import dcmread\nfrom pydicom.data import get_testdata_files","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:30.570237Z","iopub.execute_input":"2021-10-15T14:44:30.570572Z","iopub.status.idle":"2021-10-15T14:44:30.683489Z","shell.execute_reply.started":"2021-10-15T14:44:30.570537Z","shell.execute_reply":"2021-10-15T14:44:30.682895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntraining_set = []\nfor a in range(len(number_of_files_Same)):\n    for b in os.listdir(path + \"/\" + number_of_files_Same[a]):     \n        temp = []\n        for c in os.listdir(path + \"/\" + number_of_files_Same[a] + \"/\" + b):\n            temp.append(c)\n            temp = sorted(temp)\n        for d in range(len(temp)):\n            ds = dcmread(path + \"/\" + number_of_files_Same[a] + \"/\" + b + \"/\" + temp[d], force=True)\n            training_set.append(ds.pixel_array)\n\nreshape_training_set = []\nfor i in range(len(training_set)):\n    reshape_training_set.append(cv2.resize(training_set[i], (32, 32)))\nreshape_training_set = np.array(reshape_training_set)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:44:30.684901Z","iopub.execute_input":"2021-10-15T14:44:30.685405Z","iopub.status.idle":"2021-10-15T14:45:38.962967Z","shell.execute_reply.started":"2021-10-15T14:44:30.685362Z","shell.execute_reply":"2021-10-15T14:45:38.962107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del number_of_files_Same, training_set, a, b, c, d, i\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:38.964553Z","iopub.execute_input":"2021-10-15T14:45:38.964788Z","iopub.status.idle":"2021-10-15T14:45:39.077589Z","shell.execute_reply.started":"2021-10-15T14:45:38.964761Z","shell.execute_reply":"2021-10-15T14:45:39.076603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"{}{: >25}{}{: >10}{}\".format('|','Variable Name','|','Memory','|'))\nprint(\" ------------------------------------ \")\nfor var_name in dir():\n    if not var_name.startswith(\"_\") and sys.getsizeof(eval(var_name)) > 50: \n        print(\"{}{: >25}{}{: >10}{}\".format('|',var_name,'|',sys.getsizeof(eval(var_name)),'|'))","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:39.078675Z","iopub.execute_input":"2021-10-15T14:45:39.079267Z","iopub.status.idle":"2021-10-15T14:45:39.094380Z","shell.execute_reply.started":"2021-10-15T14:45:39.079202Z","shell.execute_reply":"2021-10-15T14:45:39.093495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CNN ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential, load_model       \nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:39.095738Z","iopub.execute_input":"2021-10-15T14:45:39.095983Z","iopub.status.idle":"2021-10-15T14:45:45.768287Z","shell.execute_reply.started":"2021-10-15T14:45:39.095956Z","shell.execute_reply":"2021-10-15T14:45:45.767515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = copy.deepcopy(reshape_training_set)           \ny = copy.deepcopy(new_list_label_2)\n\nX_128_128_1 = []\nfor i in range(len(X)):\n    q = np.reshape(X[i], (32,32,1))\n    X_128_128_1.append(q)\nX_128_128_1 = np.array(X_128_128_1)\nprint(X_128_128_1.shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:45.769309Z","iopub.execute_input":"2021-10-15T14:45:45.769961Z","iopub.status.idle":"2021-10-15T14:45:45.816531Z","shell.execute_reply.started":"2021-10-15T14:45:45.769925Z","shell.execute_reply":"2021-10-15T14:45:45.815943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X, reshape_training_set, new_list_label, new_list_label_2, i, q\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:45.817458Z","iopub.execute_input":"2021-10-15T14:45:45.818126Z","iopub.status.idle":"2021-10-15T14:45:45.991584Z","shell.execute_reply.started":"2021-10-15T14:45:45.818092Z","shell.execute_reply":"2021-10-15T14:45:45.990988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X_128_128_1, y, test_size=0.2, random_state=0)\nprint('x_train:', X_train.shape)\nprint('label_train:', y_train.shape)\nprint('x_test:', X_test.shape)\nprint('label_test:', y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:45.992608Z","iopub.execute_input":"2021-10-15T14:45:45.992978Z","iopub.status.idle":"2021-10-15T14:45:46.016896Z","shell.execute_reply.started":"2021-10-15T14:45:45.992949Z","shell.execute_reply":"2021-10-15T14:45:46.016103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y, X_128_128_1\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:46.017976Z","iopub.execute_input":"2021-10-15T14:45:46.018217Z","iopub.status.idle":"2021-10-15T14:45:46.210102Z","shell.execute_reply.started":"2021-10-15T14:45:46.018191Z","shell.execute_reply":"2021-10-15T14:45:46.209206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#16,(32),16,0.5,64,2 /// img:128*128 /// epoch:40, batch:all -> 0.89\n#32,32,64,0.4,128,2 /// img:128*128 /// epoch:100, batch:all -> 0.89\n#32,32,32,0.4,128,2 /// img:32*32   /// epoch:100, batch:all -> 0.84\n#32,32,32,0.4,128,2 /// img:32*32   /// epoch:50, batch:all -> 0.80\n#32,32,32,0.4,128,2 /// img:32*32   /// epoch:50, batch:150 -> 0.75\n#16,(16),16,0.4,64,2\nmodel = Sequential()\n\nmodel.add(Conv2D(32, 3, padding=\"same\", activation=\"relu\", input_shape=(32,32,1)))\nmodel.add(MaxPooling2D())\n\n#model.add(Conv2D(32, 3, padding=\"same\", activation=\"relu\"))\n#model.add(MaxPooling2D())\n\nmodel.add(Conv2D(32, 3, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPooling2D())\n\nmodel.add(Dropout(0.5))\n\nmodel.add(Flatten())\nmodel.add(Dense(64, activation=\"relu\"))\nmodel.add(Dense(2, activation=\"softmax\"))             \n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:46.211493Z","iopub.execute_input":"2021-10-15T14:45:46.212259Z","iopub.status.idle":"2021-10-15T14:45:46.354573Z","shell.execute_reply.started":"2021-10-15T14:45:46.212211Z","shell.execute_reply":"2021-10-15T14:45:46.353635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer = 'adam', loss = 'sparse_categorical_crossentropy', metrics = ['accuracy'])\nhistory = model.fit(X_train, y_train, epochs=100, batch_size=100, validation_data=(X_test,y_test)) ","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:45:46.356171Z","iopub.execute_input":"2021-10-15T14:45:46.356476Z","iopub.status.idle":"2021-10-15T14:51:20.783151Z","shell.execute_reply.started":"2021-10-15T14:45:46.356435Z","shell.execute_reply":"2021-10-15T14:51:20.782265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train, X_test, y_train, y_test, temp, Adam, Conv2D, Dense, Dropout, Flatten, MaxPooling2D, Sequential\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:51:20.784735Z","iopub.execute_input":"2021-10-15T14:51:20.784985Z","iopub.status.idle":"2021-10-15T14:51:20.976089Z","shell.execute_reply.started":"2021-10-15T14:51:20.784954Z","shell.execute_reply":"2021-10-15T14:51:20.975170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T14:51:20.977710Z","iopub.execute_input":"2021-10-15T14:51:20.978029Z","iopub.status.idle":"2021-10-15T14:51:21.013242Z","shell.execute_reply.started":"2021-10-15T14:51:20.977975Z","shell.execute_reply":"2021-10-15T14:51:21.012496Z"},"trusted":true},"execution_count":null,"outputs":[]}]}