{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom typing import Counter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T20:58:51.336774Z","iopub.execute_input":"2022-07-05T20:58:51.337383Z","iopub.status.idle":"2022-07-05T20:58:52.469778Z","shell.execute_reply.started":"2022-07-05T20:58:51.337295Z","shell.execute_reply":"2022-07-05T20:58:52.468515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_stratifiedkfold(_df, num_folds, random_state=42):\n\n    df = _df.copy(deep=True)\n\n    # split\n    cv = StratifiedKFold(n_splits=num_folds, random_state=random_state, shuffle=True)\n    for fold_index, (_, val_index) in enumerate(cv.split(df, df[\"organ\"])):\n        df.loc[val_index, \"Fold\"] = fold_index\n    df = df.astype({\"Fold\": 'int64'})\n\n    # check\n    for fold_index in range(num_folds):\n        records = df[(df[\"Fold\"] == fold_index)]\n        organ_counts = Counter(records[\"organ\"].values)\n        print(f\"fold{fold_index}: {organ_counts}\")\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:58:52.472143Z","iopub.execute_input":"2022-07-05T20:58:52.472863Z","iopub.status.idle":"2022-07-05T20:58:52.482710Z","shell.execute_reply.started":"2022-07-05T20:58:52.472816Z","shell.execute_reply":"2022-07-05T20:58:52.481968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_dir = \"../input/hubmap-organ-segmentation\"\noutput_dir = \"./\"\n\ndf = pd.read_csv(os.path.join(input_dir, \"train.csv\"))\n\nfor num_folds in [5, 10]:\n    print(f\"num_folds={num_folds}\")\n    df_out = apply_stratifiedkfold(df, num_folds)\n    df_out.to_csv(os.path.join(output_dir, f\"hubmap_organ_segmentation_{num_folds}folds.csv\"), index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:59:01.172174Z","iopub.execute_input":"2022-07-05T20:59:01.172590Z","iopub.status.idle":"2022-07-05T20:59:02.669493Z","shell.execute_reply.started":"2022-07-05T20:59:01.172559Z","shell.execute_reply":"2022-07-05T20:59:02.668664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}