{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Create the k-fold directories","metadata":{}},{"cell_type":"code","source":"import json\nfrom pathlib import Path\nimport os\nimport shutil\nimport cv2\nimport itertools\nimport numpy as np\nfrom typing import List, Dict\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.model_selection import KFold\nimport os\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:07.257725Z","iopub.execute_input":"2023-07-04T17:34:07.258134Z","iopub.status.idle":"2023-07-04T17:34:07.26389Z","shell.execute_reply.started":"2023-07-04T17:34:07.2581Z","shell.execute_reply":"2023-07-04T17:34:07.262711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = Path('/kaggle/input/hubmap-hacking-the-human-vasculature/')","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:07.621661Z","iopub.execute_input":"2023-07-04T17:34:07.622109Z","iopub.status.idle":"2023-07-04T17:34:07.62638Z","shell.execute_reply.started":"2023-07-04T17:34:07.622074Z","shell.execute_reply":"2023-07-04T17:34:07.625594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read .jsonl file and convert it to a list of dicts\n# The dicts contain IDs, class names and segmentation masks\n# from https://www.kaggle.com/code/leonidkulyk/eda-hubmap-hhv-interactive-annotations\nwith open('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl', 'r') as json_file:\n    json_list = list(json_file)\n    \ntiles_dicts = []\nfor json_str in json_list:\n    tiles_dicts.append(json.loads(json_str))","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:08.296091Z","iopub.execute_input":"2023-07-04T17:34:08.297565Z","iopub.status.idle":"2023-07-04T17:34:13.131766Z","shell.execute_reply.started":"2023-07-04T17:34:08.297504Z","shell.execute_reply":"2023-07-04T17:34:13.130604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a conversion between class name and number\nid_dict = {'blood_vessel': 0, 'glomerulus': 1, 'unsure': 2}","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:13.133877Z","iopub.execute_input":"2023-07-04T17:34:13.134258Z","iopub.status.idle":"2023-07-04T17:34:13.138812Z","shell.execute_reply.started":"2023-07-04T17:34:13.134226Z","shell.execute_reply":"2023-07-04T17:34:13.137711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to copy images and transform labels to \n# coco formatted .txt files\ndef tile_to_coco(tile: List[Dict], output_folder: Path):\n    tile_id = tile['id']    \n    \n    # Copy image\n    shutil.copyfile(DATA_DIR / f'train/{tile_id}.tif', output_folder / f'{tile_id}.tif')\n    \n    # Create text file and write formatted labels to it\n    with open(output_folder / f'{tile_id}.txt', 'w') as text_file:\n        for annotation in tile['annotations']:\n            \n            class_id = id_dict[annotation['type']]\n            flat_mask_polygon = list(itertools.chain(*annotation['coordinates'][0]))\n            # Divide by 512 because coco labels expect positions between 0 and 1\n            # not pixel indices\n            array = np.array(flat_mask_polygon)/512.\n            text_file.write(f'{class_id} {\" \".join(map(str, array))}\\n')\n            \n        ","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:13.140132Z","iopub.execute_input":"2023-07-04T17:34:13.140498Z","iopub.status.idle":"2023-07-04T17:34:13.158421Z","shell.execute_reply.started":"2023-07-04T17:34:13.140467Z","shell.execute_reply":"2023-07-04T17:34:13.157179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Folds directories","metadata":{}},{"cell_type":"code","source":"# Assuming k=5 for 5-fold cross-validation\nk = 5\n\n# Create the k-fold directories\nfor i in range(k):\n    fold_dir = f'/kaggle/working/fold{i}/'\n    os.makedirs(fold_dir)\n    os.makedirs(fold_dir + 'train/')\n    os.makedirs(fold_dir + 'valid/')","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:13.160991Z","iopub.execute_input":"2023-07-04T17:34:13.161392Z","iopub.status.idle":"2023-07-04T17:34:13.172292Z","shell.execute_reply.started":"2023-07-04T17:34:13.161359Z","shell.execute_reply":"2023-07-04T17:34:13.170875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split Data","metadata":{}},{"cell_type":"code","source":"# Perform k-fold cross-validation\nkf = KFold(n_splits=k, random_state=42, shuffle=True)\nfor fold_index, (train_idx, valid_idx) in enumerate(kf.split(tiles_dicts)):\n    train_dicts = [tiles_dicts[i] for i in train_idx]\n    valid_dicts = [tiles_dicts[i] for i in valid_idx]\n\n    fold_dir = f'/kaggle/working/fold{fold_index}/'\n\n    # Copy the train and valid data to the corresponding fold directory\n    for train_dict in train_dicts:\n        tile_to_coco(train_dict, Path(fold_dir + 'train/'))\n\n    for valid_dict in valid_dicts:\n        tile_to_coco(valid_dict, Path(fold_dir + 'valid/'))\n\n    # Create the hubmap-coco.yaml file for each fold\n    yaml_text = f\"\"\"\n    # HuBMAP - Hacking the Human Vasculature dataset \n    # https://www.kaggle.com/competitions/hubmap-hacking-the-human-vasculature\n\n    # train and val data as 1) directory: path/images/, 2) file: path/images.txt, or 3) list: [path1/images/, path2/images/]\n    train: /kaggle/working/fold{fold_index}/train/\n    val: /kaggle/working/fold{fold_index}/valid/\n\n    # class names\n    names: \n      0: blood_vessel\n      1: glomerulus\n      2: unsure\n    \"\"\"\n\n    with open(fold_dir + 'hubmap-coco.yaml', 'w') as text_file:\n        text_file.write(yaml_text)","metadata":{"execution":{"iopub.status.busy":"2023-07-04T17:34:25.033915Z","iopub.execute_input":"2023-07-04T17:34:25.034469Z","iopub.status.idle":"2023-07-04T17:35:28.947857Z","shell.execute_reply.started":"2023-07-04T17:34:25.03443Z","shell.execute_reply":"2023-07-04T17:35:28.946256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Folders","metadata":{}},{"cell_type":"code","source":"!zip -r fold0.zip fold0\n!rm -rf fold0\n\n!zip -r fold1.zip fold1\n!rm -rf fold1\n\n!zip -r fold2.zip fold2\n!rm -rf fold2\n\n!zip -r fold3.zip fold3\n!rm -rf fold3\n\n!zip -r fold4.zip fold4\n!rm -rf fold4","metadata":{},"execution_count":null,"outputs":[]}]}