{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install iterative-stratification","metadata":{"execution":{"iopub.status.busy":"2024-08-19T08:17:24.035888Z","iopub.execute_input":"2024-08-19T08:17:24.036307Z","iopub.status.idle":"2024-08-19T08:17:40.963947Z","shell.execute_reply.started":"2024-08-19T08:17:24.036276Z","shell.execute_reply":"2024-08-19T08:17:40.962473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-19T08:22:35.250278Z","iopub.execute_input":"2024-08-19T08:22:35.250681Z","iopub.status.idle":"2024-08-19T08:22:35.256299Z","shell.execute_reply.started":"2024-08-19T08:22:35.250651Z","shell.execute_reply":"2024-08-19T08:22:35.255004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ndf = df.fillna('Unknown')","metadata":{"execution":{"iopub.status.busy":"2024-08-19T08:23:19.030484Z","iopub.execute_input":"2024-08-19T08:23:19.030927Z","iopub.status.idle":"2024-08-19T08:23:19.065160Z","shell.execute_reply.started":"2024-08-19T08:23:19.030893Z","shell.execute_reply":"2024-08-19T08:23:19.063904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from iterstrat.ml_stratifiers import MultilabelStratifiedKFold\nimport numpy as np\n\nmskf = MultilabelStratifiedKFold(n_splits=5, shuffle=True, random_state=0)\n\nfold = 0\nfor train_index, test_index in mskf.split(df, df.iloc[:,1:]):\n    df.loc[test_index, 'fold'] = fold\n    fold += 1","metadata":{"execution":{"iopub.status.busy":"2024-08-19T08:23:58.303057Z","iopub.execute_input":"2024-08-19T08:23:58.303472Z","iopub.status.idle":"2024-08-19T08:23:58.423700Z","shell.execute_reply.started":"2024-08-19T08:23:58.303439Z","shell.execute_reply":"2024-08-19T08:23:58.422446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['fold'] = df['fold'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T08:24:13.318044Z","iopub.execute_input":"2024-08-19T08:24:13.318542Z","iopub.status.idle":"2024-08-19T08:24:13.325374Z","shell.execute_reply.started":"2024-08-19T08:24:13.318499Z","shell.execute_reply":"2024-08-19T08:24:13.323966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['study_id', 'fold']].to_csv('5folds.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-19T08:24:50.559490Z","iopub.execute_input":"2024-08-19T08:24:50.559953Z","iopub.status.idle":"2024-08-19T08:24:50.574358Z","shell.execute_reply.started":"2024-08-19T08:24:50.559915Z","shell.execute_reply":"2024-08-19T08:24:50.573094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}