{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Adding Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn import model_selection","metadata":{"execution":{"iopub.status.busy":"2021-09-30T16:40:15.840972Z","iopub.execute_input":"2021-09-30T16:40:15.841323Z","iopub.status.idle":"2021-09-30T16:40:16.175399Z","shell.execute_reply.started":"2021-09-30T16:40:15.841267Z","shell.execute_reply":"2021-09-30T16:40:16.174390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Read the data","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"../input/g2net-gravitational-wave-detection/training_labels.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-30T16:40:16.181560Z","iopub.execute_input":"2021-09-30T16:40:16.182069Z","iopub.status.idle":"2021-09-30T16:40:16.568802Z","shell.execute_reply.started":"2021-09-30T16:40:16.182030Z","shell.execute_reply":"2021-09-30T16:40:16.567988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Set a new column in the data frame to split the data onto k folds\n","metadata":{}},{"cell_type":"code","source":"data[\"k_fold_splitting\"] = -1","metadata":{"execution":{"iopub.status.busy":"2021-09-30T16:40:16.570140Z","iopub.execute_input":"2021-09-30T16:40:16.571010Z","iopub.status.idle":"2021-09-30T16:40:16.577693Z","shell.execute_reply.started":"2021-09-30T16:40:16.570973Z","shell.execute_reply":"2021-09-30T16:40:16.576829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"n_splits : No of splits we want to make of the data\n\nshuffle : Whether to shuffle each class’s samples before splitting into batches.\n\nrandom_state: When shuffle is True, random_state affects the ordering of the indices, which controls the randomness of each fold for each class. \n\n- **Pass an int for reproducible output across multiple function calls**","metadata":{}},{"cell_type":"code","source":"k_fold = model_selection.StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\nfor fold, (_, value) in enumerate(k_fold.split(X=data, y=data.target.values)):\n    data.loc[value, 'k_fold_splitting'] = fold","metadata":{"execution":{"iopub.status.busy":"2021-09-30T16:40:16.579798Z","iopub.execute_input":"2021-09-30T16:40:16.580205Z","iopub.status.idle":"2021-09-30T16:40:16.763966Z","shell.execute_reply.started":"2021-09-30T16:40:16.580170Z","shell.execute_reply":"2021-09-30T16:40:16.763141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A new column **k_fold_splitting** is created which contains the fold of each row.","metadata":{}},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-30T16:40:39.791312Z","iopub.execute_input":"2021-09-30T16:40:39.791609Z","iopub.status.idle":"2021-09-30T16:40:39.802374Z","shell.execute_reply.started":"2021-09-30T16:40:39.791578Z","shell.execute_reply":"2021-09-30T16:40:39.801686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Converting the data frame to a csv file.","metadata":{}},{"cell_type":"code","source":"data.to_csv(\"k_fold_train_folds.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-30T16:40:16.780545Z","iopub.execute_input":"2021-09-30T16:40:16.780860Z","iopub.status.idle":"2021-09-30T16:40:18.017386Z","shell.execute_reply.started":"2021-09-30T16:40:16.780798Z","shell.execute_reply":"2021-09-30T16:40:18.016407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}