{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\nimport random\nfrom sklearn.model_selection import StratifiedKFold\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.style.use('seaborn-talk')\nimport pytorch_lightning as pl ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-18T13:57:23.692459Z","iopub.execute_input":"2021-08-18T13:57:23.692986Z","iopub.status.idle":"2021-08-18T13:57:24.813124Z","shell.execute_reply.started":"2021-08-18T13:57:23.692957Z","shell.execute_reply":"2021-08-18T13:57:24.812091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 42\n\npl.utilities.seed.seed_everything(SEED, workers=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-18T13:57:34.270094Z","iopub.execute_input":"2021-08-18T13:57:34.270598Z","iopub.status.idle":"2021-08-18T13:57:34.283170Z","shell.execute_reply.started":"2021-08-18T13:57:34.270560Z","shell.execute_reply":"2021-08-18T13:57:34.282463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\", dtype={'BraTS21ID': str})\nprint(df.shape)\ndf = df[~df.BraTS21ID.isin(['00109', '00123', '00709'])].reset_index()\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-18T13:59:27.246257Z","iopub.execute_input":"2021-08-18T13:59:27.247425Z","iopub.status.idle":"2021-08-18T13:59:27.260638Z","shell.execute_reply.started":"2021-08-18T13:59:27.247381Z","shell.execute_reply":"2021-08-18T13:59:27.259471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-08-18T13:59:28.966867Z","iopub.execute_input":"2021-08-18T13:59:28.967247Z","iopub.status.idle":"2021-08-18T13:59:28.977576Z","shell.execute_reply.started":"2021-08-18T13:59:28.967216Z","shell.execute_reply":"2021-08-18T13:59:28.976701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nskf = StratifiedKFold(random_state=SEED, shuffle=True)\n\ndf['fold'] = -1\nfor fold, (train_i, val_i) in enumerate(skf.split(df, df.MGMT_value)):\n    df.loc[val_i, 'fold'] = fold\n    \ndf.fold.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-08-18T13:59:34.935322Z","iopub.execute_input":"2021-08-18T13:59:34.935720Z","iopub.status.idle":"2021-08-18T13:59:34.950151Z","shell.execute_reply.started":"2021-08-18T13:59:34.935692Z","shell.execute_reply":"2021-08-18T13:59:34.949521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-18T13:59:44.980356Z","iopub.execute_input":"2021-08-18T13:59:44.981048Z","iopub.status.idle":"2021-08-18T13:59:44.991358Z","shell.execute_reply.started":"2021-08-18T13:59:44.981014Z","shell.execute_reply":"2021-08-18T13:59:44.990671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('brain_tumor_kfold.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-18T13:59:46.107174Z","iopub.execute_input":"2021-08-18T13:59:46.107642Z","iopub.status.idle":"2021-08-18T13:59:46.116136Z","shell.execute_reply.started":"2021-08-18T13:59:46.107616Z","shell.execute_reply":"2021-08-18T13:59:46.115336Z"},"trusted":true},"execution_count":null,"outputs":[]}]}