{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob\nimport pandas as pd\nimport os\nimport pydicom\nfrom tqdm import tqdm\n\ndef get_metadata(dcm_dir):\n    dcm_paths = sorted(glob.glob(os.path.join(dcm_dir, \"*.dcm\")), key=lambda x: int(os.path.basename(x).replace(\".dcm\", \"\").replace(\"Image-\", \"\")))\n    \n    df = {}\n    for dcm_path in dcm_paths:\n        img = pydicom.dcmread(str(dcm_path))\n        for k in img:\n            if k.name == \"Percent Phase Field of View\":\n                df['Percent Phase Field of View'] = k.value\n            elif k.name == \"Echo Train Length\":\n                df['Echo Train Length'] = k.value\n            elif k.name == \"Series Description\":\n                df['Series Description'] = k.value\n\n    df[\"shape\"] = len(dcm_paths)\n\n    new_df = {}\n    for k, v in df.items():\n        if k != \"Series Description\":\n            new_df[df[\"Series Description\"] + \"_\" + k] = v \n    \n    return new_df","metadata":{"execution":{"iopub.status.busy":"2021-09-28T13:37:26.204608Z","iopub.execute_input":"2021-09-28T13:37:26.205097Z","iopub.status.idle":"2021-09-28T13:37:26.527232Z","shell.execute_reply.started":"2021-09-28T13:37:26.204995Z","shell.execute_reply":"2021-09-28T13:37:26.526153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv',\n                     dtype={'BraTS21ID': object})\nX = []\n\nfor i in tqdm(range(len(labels))):\n    idt = labels.loc[i,'BraTS21ID']\n\n    flair = get_metadata(os.path.join('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train', idt, 'FLAIR'))\n    t2w = get_metadata(os.path.join('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train', idt, 'T2w'))\n        \n    new_df = dict(**flair, **t2w)\n    new_df[\"target\"] = labels.loc[i,'MGMT_value']\n\n    X.append(new_df)\n\nX = pd.DataFrame(X)\nX.to_csv(\"X_train.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T13:37:26.528717Z","iopub.execute_input":"2021-09-28T13:37:26.529031Z","iopub.status.idle":"2021-09-28T14:03:46.701281Z","shell.execute_reply.started":"2021-09-28T13:37:26.529Z","shell.execute_reply":"2021-09-28T14:03:46.700229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv',\n                     dtype={'BraTS21ID': object})\nX = []\n\nfor i in tqdm(range(len(labels))):\n    idt = labels.loc[i,'BraTS21ID']\n\n    flair = get_metadata(os.path.join('../input/rsna-miccai-brain-tumor-radiogenomic-classification/test', idt, 'FLAIR'))\n    t2w = get_metadata(os.path.join('../input/rsna-miccai-brain-tumor-radiogenomic-classification/test', idt, 'T2w'))\n        \n    new_df = dict(**flair, **t2w)\n    new_df[\"target\"] = labels.loc[i,'MGMT_value']\n\n    X.append(new_df)\n\nX = pd.DataFrame(X)\nX.to_csv(\"X_test.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T14:03:46.704008Z","iopub.execute_input":"2021-09-28T14:03:46.704444Z","iopub.status.idle":"2021-09-28T14:08:04.861733Z","shell.execute_reply.started":"2021-09-28T14:03:46.704401Z","shell.execute_reply":"2021-09-28T14:08:04.860815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import log_loss\nimport numpy as np\n\nX = pd.read_csv(\"X_train.csv\", usecols=['T2w_Percent Phase Field of View',\n                                        'FLAIR_Echo Train Length',\n                                        'T2w_shape',\n                                        'target'])\nX.fillna(0, inplace=True)\ny = X[\"target\"].values\nX.drop(\"target\", axis=1, inplace=True)\nX = X.values\n\nX_test = pd.read_csv(\"X_test.csv\", usecols=['T2w_Percent Phase Field of View',\n                                             'FLAIR_Echo Train Length',\n                                             'T2w_shape'])\nX_test.fillna(0, inplace=True)\nX_test = X_test.values\n\no = []\no2 = []\npredicted_test = np.zeros(X_test.shape[0])\nfor fold, (train_index, val_index) in enumerate(StratifiedKFold(n_splits=200).split(X, y)):\n    X_train, X_val = X[train_index], X[val_index]\n    y_train, y_val = y[train_index], y[val_index]\n\n    regr = LogisticRegression()\n    regr.fit(X_train, y_train)\n    \n    y_pred = regr.predict_proba(X_val)[...,1]\n    \n    auc = roc_auc_score(y_val, y_pred)\n    val_loss = log_loss(y_val, y_pred)\n    \n    o.append(auc)\n    o2.append(val_loss)\n    \n    predicted_test += regr.predict_proba(X_test)[...,1]\n\nprint(\"Loss\", np.mean(o2), np.std(o2))\nprint(\"AUC\", np.mean(o), np.std(o))\n\npredicted_test /= 200","metadata":{"execution":{"iopub.status.busy":"2021-09-28T14:09:06.421702Z","iopub.execute_input":"2021-09-28T14:09:06.422124Z","iopub.status.idle":"2021-09-28T14:09:08.478028Z","shell.execute_reply.started":"2021-09-28T14:09:06.422088Z","shell.execute_reply":"2021-09-28T14:09:08.477193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv',\n                     dtype={'BraTS21ID': object})\nlabels[\"MGMT_value\"] = predicted_test\nlabels.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T14:09:11.823575Z","iopub.execute_input":"2021-09-28T14:09:11.823978Z","iopub.status.idle":"2021-09-28T14:09:11.835685Z","shell.execute_reply.started":"2021-09-28T14:09:11.823944Z","shell.execute_reply":"2021-09-28T14:09:11.834659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels[\"MGMT_value\"].tolist()","metadata":{"execution":{"iopub.status.busy":"2021-09-28T14:09:13.769447Z","iopub.execute_input":"2021-09-28T14:09:13.769792Z","iopub.status.idle":"2021-09-28T14:09:13.777466Z","shell.execute_reply.started":"2021-09-28T14:09:13.769765Z","shell.execute_reply":"2021-09-28T14:09:13.776443Z"},"trusted":true},"execution_count":null,"outputs":[]}]}