{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Based on [this discussion](https://www.kaggle.com/competitions/feedback-prize-english-language-learning/discussion/368437), MultilabelStratifiedKFold **does not work** with inputs other than binary inputs.   \n\nTherefore, it is necessary to follow these steps:　　\n1. Convert the labels into binary inputs using One-Hot encoding.\n2. Apply MultilabelStratifiedKFold.\n\nBelow is the experiment.  ","metadata":{}},{"cell_type":"code","source":"!pip install iterative-stratification -qqq","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:01:38.037486Z","iopub.execute_input":"2024-09-14T11:01:38.038030Z","iopub.status.idle":"2024-09-14T11:01:52.553698Z","shell.execute_reply.started":"2024-09-14T11:01:38.037980Z","shell.execute_reply":"2024-09-14T11:01:52.552243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\npd.set_option('display.max_columns', 1000)","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:01:52.556570Z","iopub.execute_input":"2024-09-14T11:01:52.557677Z","iopub.status.idle":"2024-09-14T11:01:52.972148Z","shell.execute_reply.started":"2024-09-14T11:01:52.557594Z","shell.execute_reply":"2024-09-14T11:01:52.971188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.Apply MultilabelStratifiedKFold as is.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ndf = df.fillna('Unknown')\nlabel_cols = df.iloc[:,1:].columns","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:39.457216Z","iopub.execute_input":"2024-09-14T11:11:39.457691Z","iopub.status.idle":"2024-09-14T11:11:39.486795Z","shell.execute_reply.started":"2024-09-14T11:11:39.457638Z","shell.execute_reply":"2024-09-14T11:11:39.485804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from iterstrat.ml_stratifiers import MultilabelStratifiedKFold\nmskf = MultilabelStratifiedKFold(n_splits=5, shuffle=True, random_state=0)\nfold = 0\nfor train_index, test_index in mskf.split(df, df.iloc[:,1:]):\n    df.loc[test_index, 'fold'] = fold\n    fold += 1","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:39.627117Z","iopub.execute_input":"2024-09-14T11:11:39.627562Z","iopub.status.idle":"2024-09-14T11:11:39.716999Z","shell.execute_reply.started":"2024-09-14T11:11:39.627521Z","shell.execute_reply":"2024-09-14T11:11:39.715845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fold distribution","metadata":{}},{"cell_type":"code","source":"one_hot_cols = pd.get_dummies(df, columns=label_cols).iloc[:,2:].columns","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:39.918557Z","iopub.execute_input":"2024-09-14T11:11:39.919872Z","iopub.status.idle":"2024-09-14T11:11:39.960720Z","shell.execute_reply.started":"2024-09-14T11:11:39.919805Z","shell.execute_reply":"2024-09-14T11:11:39.959297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.get_dummies(df, columns=label_cols).groupby(\"fold\")[one_hot_cols].sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:40.027415Z","iopub.execute_input":"2024-09-14T11:11:40.027877Z","iopub.status.idle":"2024-09-14T11:11:40.140074Z","shell.execute_reply.started":"2024-09-14T11:11:40.027836Z","shell.execute_reply":"2024-09-14T11:11:40.138756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:40.169618Z","iopub.execute_input":"2024-09-14T11:11:40.170072Z","iopub.status.idle":"2024-09-14T11:11:40.206832Z","shell.execute_reply.started":"2024-09-14T11:11:40.170035Z","shell.execute_reply":"2024-09-14T11:11:40.205587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.After applying One-Hot encoding to the labels, MultilabelStratifiedKFold is applied.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification/train.csv')\ndf = df.fillna('Unknown')\nlabel_cols = df.iloc[:,1:].columns","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:46.404385Z","iopub.execute_input":"2024-09-14T11:11:46.405213Z","iopub.status.idle":"2024-09-14T11:11:46.434172Z","shell.execute_reply.started":"2024-09-14T11:11:46.405169Z","shell.execute_reply":"2024-09-14T11:11:46.433006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mskf = MultilabelStratifiedKFold(n_splits=5, shuffle=True, random_state=0)\n# apply one-hot encoding\ny = pd.get_dummies(df, columns=df.columns[1:])\nfold = 0\nfor train_index, test_index in mskf.split(df, y):\n    df.loc[test_index, 'fold'] = fold\n    fold += 1","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:46.545228Z","iopub.execute_input":"2024-09-14T11:11:46.545630Z","iopub.status.idle":"2024-09-14T11:11:46.672602Z","shell.execute_reply.started":"2024-09-14T11:11:46.545593Z","shell.execute_reply":"2024-09-14T11:11:46.671526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fold distribution","metadata":{}},{"cell_type":"code","source":"pd.get_dummies(df, columns=label_cols).groupby(\"fold\")[one_hot_cols].sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:46.850356Z","iopub.execute_input":"2024-09-14T11:11:46.851355Z","iopub.status.idle":"2024-09-14T11:11:46.928866Z","shell.execute_reply.started":"2024-09-14T11:11:46.851309Z","shell.execute_reply":"2024-09-14T11:11:46.927737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-14T11:11:52.043565Z","iopub.execute_input":"2024-09-14T11:11:52.044021Z","iopub.status.idle":"2024-09-14T11:11:52.072055Z","shell.execute_reply.started":"2024-09-14T11:11:52.043981Z","shell.execute_reply":"2024-09-14T11:11:52.070835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By converting the correct labels to binary, it was confirmed that MultilabelStratifiedKFold is working properly.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}