{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Notebook demonstrating Group Stratified KFold Cross Vaidation technique on train dataset.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ndata_dir = '/kaggle/input/siim-isic-melanoma-classification/'\nprint(os.listdir(data_dir))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-04-25T04:31:31.328354Z","iopub.execute_input":"2023-04-25T04:31:31.328818Z","iopub.status.idle":"2023-04-25T04:31:31.365789Z","shell.execute_reply.started":"2023-04-25T04:31:31.328776Z","shell.execute_reply":"2023-04-25T04:31:31.364387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading the train.csv dataset","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:31:59.983655Z","iopub.execute_input":"2023-04-25T04:31:59.984097Z","iopub.status.idle":"2023-04-25T04:32:00.071026Z","shell.execute_reply.started":"2023-04-25T04:31:59.984056Z","shell.execute_reply":"2023-04-25T04:32:00.069978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:32:00.313824Z","iopub.execute_input":"2023-04-25T04:32:00.316449Z","iopub.status.idle":"2023-04-25T04:32:00.349468Z","shell.execute_reply.started":"2023-04-25T04:32:00.316402Z","shell.execute_reply":"2023-04-25T04:32:00.348141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Importing model_selection from sklearn\nfrom sklearn import model_selection","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:32:16.106114Z","iopub.execute_input":"2023-04-25T04:32:16.106581Z","iopub.status.idle":"2023-04-25T04:32:16.732931Z","shell.execute_reply.started":"2023-04-25T04:32:16.106540Z","shell.execute_reply":"2023-04-25T04:32:16.731689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assigning a column dedicated to groupstratifiedkfold in the dataset: This is the column that gets populated later on with the fold-number.\ntrain_data['gkfold'] = -1","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:33:13.090808Z","iopub.execute_input":"2023-04-25T04:33:13.091257Z","iopub.status.idle":"2023-04-25T04:33:13.101576Z","shell.execute_reply.started":"2023-04-25T04:33:13.091217Z","shell.execute_reply":"2023-04-25T04:33:13.100110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Randomizing the dataset\nrandomized_data = train_data.sample(frac=1).reset_index(drop=True)\nrandomized_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:33:29.828103Z","iopub.execute_input":"2023-04-25T04:33:29.828535Z","iopub.status.idle":"2023-04-25T04:33:29.863512Z","shell.execute_reply.started":"2023-04-25T04:33:29.828499Z","shell.execute_reply":"2023-04-25T04:33:29.861983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the dataset shape\ntrain_data.shape, randomized_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:33:45.067138Z","iopub.execute_input":"2023-04-25T04:33:45.067558Z","iopub.status.idle":"2023-04-25T04:33:45.075617Z","shell.execute_reply.started":"2023-04-25T04:33:45.067522Z","shell.execute_reply":"2023-04-25T04:33:45.074122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#initializing groupkfold instance\ngkf = model_selection.GroupKFold(n_splits=5)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:35:18.823436Z","iopub.execute_input":"2023-04-25T04:35:18.823894Z","iopub.status.idle":"2023-04-25T04:35:18.829866Z","shell.execute_reply.started":"2023-04-25T04:35:18.823850Z","shell.execute_reply":"2023-04-25T04:35:18.828532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# In Group stratified Kfold cross validation, we can assign a groups value in order to uniquely distribute the samples according to this columns unique values.\nfor f,(t,v) in enumerate(gkf.split(X=randomized_data, y = randomized_data.benign_malignant.values, groups=randomized_data.patient_id.values)):\n    randomized_data.loc[v,'gkfold'] = f","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:35:20.059269Z","iopub.execute_input":"2023-04-25T04:35:20.059661Z","iopub.status.idle":"2023-04-25T04:35:20.108014Z","shell.execute_reply.started":"2023-04-25T04:35:20.059628Z","shell.execute_reply":"2023-04-25T04:35:20.106974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data[randomized_data['gkfold'] == 1]['patient_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:35:22.416985Z","iopub.execute_input":"2023-04-25T04:35:22.417459Z","iopub.status.idle":"2023-04-25T04:35:22.434835Z","shell.execute_reply.started":"2023-04-25T04:35:22.417415Z","shell.execute_reply":"2023-04-25T04:35:22.433634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data['patient_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:35:29.578860Z","iopub.execute_input":"2023-04-25T04:35:29.579279Z","iopub.status.idle":"2023-04-25T04:35:29.593543Z","shell.execute_reply.started":"2023-04-25T04:35:29.579242Z","shell.execute_reply":"2023-04-25T04:35:29.591926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that in the above fold column, the distribution is a little skewed and this is why we need Stratified K fold on top of the group kfold cross validation.","metadata":{}},{"cell_type":"code","source":"# Initializing Stratified Kfold crioss validation instance\nskf = model_selection.StratifiedKFold(n_splits = 5)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:37:06.651484Z","iopub.execute_input":"2023-04-25T04:37:06.651932Z","iopub.status.idle":"2023-04-25T04:37:06.658236Z","shell.execute_reply.started":"2023-04-25T04:37:06.651891Z","shell.execute_reply":"2023-04-25T04:37:06.656281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #Again instantiating kfold column to -1 which gets populated later on\nrandomized_data['kfold'] = -1","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:37:34.506401Z","iopub.execute_input":"2023-04-25T04:37:34.506804Z","iopub.status.idle":"2023-04-25T04:37:34.512541Z","shell.execute_reply.started":"2023-04-25T04:37:34.506769Z","shell.execute_reply":"2023-04-25T04:37:34.511416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f, (t,v) in enumerate(skf.split(X= randomized_data, y = randomized_data.gkfold.values)):\n    randomized_data.loc[v,'kfold'] = f","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:37:37.001151Z","iopub.execute_input":"2023-04-25T04:37:37.002267Z","iopub.status.idle":"2023-04-25T04:37:37.020982Z","shell.execute_reply.started":"2023-04-25T04:37:37.002221Z","shell.execute_reply":"2023-04-25T04:37:37.020082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:37:47.637849Z","iopub.execute_input":"2023-04-25T04:37:47.638765Z","iopub.status.idle":"2023-04-25T04:37:47.655957Z","shell.execute_reply.started":"2023-04-25T04:37:47.638718Z","shell.execute_reply":"2023-04-25T04:37:47.654593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the stratified kfold sample distribution.\nrandomized_data.kfold.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:37:49.423461Z","iopub.execute_input":"2023-04-25T04:37:49.423899Z","iopub.status.idle":"2023-04-25T04:37:49.433915Z","shell.execute_reply.started":"2023-04-25T04:37:49.423852Z","shell.execute_reply":"2023-04-25T04:37:49.432760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the stratified kfold sample distribution with respect to the group patient_id\nrandomized_data[randomized_data['kfold'] == 1]['patient_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:38:35.718291Z","iopub.execute_input":"2023-04-25T04:38:35.719283Z","iopub.status.idle":"2023-04-25T04:38:35.735077Z","shell.execute_reply.started":"2023-04-25T04:38:35.719237Z","shell.execute_reply":"2023-04-25T04:38:35.733895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Trying out the sklearn group stratified kfold cross validation and comparing tghe results\n\ngskfold = model_selection.StratifiedGroupKFold(n_splits=5)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:39:43.439389Z","iopub.execute_input":"2023-04-25T04:39:43.439802Z","iopub.status.idle":"2023-04-25T04:39:43.445493Z","shell.execute_reply.started":"2023-04-25T04:39:43.439762Z","shell.execute_reply":"2023-04-25T04:39:43.444226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data['gkfold_sk'] = -1","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:39:44.565683Z","iopub.execute_input":"2023-04-25T04:39:44.566117Z","iopub.status.idle":"2023-04-25T04:39:44.572883Z","shell.execute_reply.started":"2023-04-25T04:39:44.566075Z","shell.execute_reply":"2023-04-25T04:39:44.571349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f,(t,v) in enumerate(gskfold.split(X=randomized_data, y=randomized_data.benign_malignant.values, groups=randomized_data.patient_id.values)):\n    randomized_data.loc[v,'gkfold_sk'] = f","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:39:45.358363Z","iopub.execute_input":"2023-04-25T04:39:45.358795Z","iopub.status.idle":"2023-04-25T04:39:46.290894Z","shell.execute_reply.started":"2023-04-25T04:39:45.358754Z","shell.execute_reply":"2023-04-25T04:39:46.289633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:39:46.527594Z","iopub.execute_input":"2023-04-25T04:39:46.528894Z","iopub.status.idle":"2023-04-25T04:39:46.546028Z","shell.execute_reply.started":"2023-04-25T04:39:46.528846Z","shell.execute_reply":"2023-04-25T04:39:46.544764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data['gkfold_sk'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:39:49.643892Z","iopub.execute_input":"2023-04-25T04:39:49.644353Z","iopub.status.idle":"2023-04-25T04:39:49.653649Z","shell.execute_reply.started":"2023-04-25T04:39:49.644312Z","shell.execute_reply":"2023-04-25T04:39:49.652417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data[randomized_data['gkfold_sk'] == 1]['patient_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:39:50.421388Z","iopub.execute_input":"2023-04-25T04:39:50.422492Z","iopub.status.idle":"2023-04-25T04:39:50.436678Z","shell.execute_reply.started":"2023-04-25T04:39:50.422445Z","shell.execute_reply":"2023-04-25T04:39:50.435257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data['gkfold_sk2'] = -1","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:40:38.585420Z","iopub.execute_input":"2023-04-25T04:40:38.585817Z","iopub.status.idle":"2023-04-25T04:40:38.592272Z","shell.execute_reply.started":"2023-04-25T04:40:38.585782Z","shell.execute_reply":"2023-04-25T04:40:38.590929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f,(t,v) in enumerate(gskfold.split(X=randomized_data, y=randomized_data.gkfold.values)):\n    randomized_data.loc[v,'gkfold_sk2'] = f","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:40:52.354117Z","iopub.execute_input":"2023-04-25T04:40:52.354583Z","iopub.status.idle":"2023-04-25T04:40:52.369462Z","shell.execute_reply.started":"2023-04-25T04:40:52.354542Z","shell.execute_reply":"2023-04-25T04:40:52.368160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"randomized_data['gkfold_sk2'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T04:40:54.265351Z","iopub.execute_input":"2023-04-25T04:40:54.265776Z","iopub.status.idle":"2023-04-25T04:40:54.276822Z","shell.execute_reply.started":"2023-04-25T04:40:54.265727Z","shell.execute_reply":"2023-04-25T04:40:54.275116Z"},"trusted":true},"execution_count":null,"outputs":[]}]}