{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Very Quick EDA","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"from typing import List, Union\nimport os\n\nimport tifffile\nimport pandas as pd\nimport numpy as np\n\nfrom matplotlib import pyplot as plt\n\nimport json\nimport cv2\n\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:34:54.162323Z","iopub.execute_input":"2023-05-24T11:34:54.162943Z","iopub.status.idle":"2023-05-24T11:34:54.168848Z","shell.execute_reply.started":"2023-05-24T11:34:54.162910Z","shell.execute_reply":"2023-05-24T11:34:54.167562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DATA_DIR = \"/kaggle/input/hubmap-hacking-the-human-vasculature\"\nTRAIN_IMAGE_DIR = os.path.join(BASE_DATA_DIR, \"train\")\nTRAIN_MASKS_FILE = os.path.join(BASE_DATA_DIR, \"polygons.jsonl\")\nTILE_METADATA_FILE = os.path.join(BASE_DATA_DIR, \"tile_meta.csv\")\nWSI_METADATA_FILE = os.path.join(BASE_DATA_DIR, \"wsi_meta.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:14:42.407265Z","iopub.execute_input":"2023-05-24T11:14:42.407637Z","iopub.status.idle":"2023-05-24T11:14:42.413892Z","shell.execute_reply.started":"2023-05-24T11:14:42.407610Z","shell.execute_reply":"2023-05-24T11:14:42.412316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def has_ext(file_name: str, extensions: Union[str, List[str]]) -> bool:\n    if not isinstance(extensions, (str, list)):\n        raise ValueError(\"Argument extensions must be either string or list of strings\")\n    if isinstance(extensions, str):\n        extensions = [extensions]\n    extensions = set(map(str.lower, extensions))\n\n    name, ext = os.path.splitext(file_name)\n    return ext.lower() in extensions\n\n\ndef find_in_dir_with_ext(dir_path: str, extensions: Union[str, List[str]]) -> List[str]:\n    return [os.path.join(dir_path, file_name) for file_name in sorted(os.listdir(dir_path)) if has_ext(file_name, extensions)]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:13:25.656219Z","iopub.execute_input":"2023-05-24T11:13:25.656672Z","iopub.status.idle":"2023-05-24T11:13:25.664648Z","shell.execute_reply.started":"2023-05-24T11:13:25.656635Z","shell.execute_reply":"2023-05-24T11:13:25.663493Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file_list = find_in_dir_with_ext(TRAIN_IMAGE_DIR, [\".tiff\", \".tif\"])\nprint(len(train_file_list))\nprint(train_file_list[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:26:53.430680Z","iopub.execute_input":"2023-05-24T11:26:53.431035Z","iopub.status.idle":"2023-05-24T11:26:53.745428Z","shell.execute_reply.started":"2023-05-24T11:26:53.431006Z","shell.execute_reply":"2023-05-24T11:26:53.744240Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tiff image metadata\nDidn't see much insightful info besides lzw compression","metadata":{}},{"cell_type":"code","source":"tif_path = train_file_list[0]\ntif = tifffile.TiffFile(tif_path)\n\ntif_data = tifffile.imread(tif_path)\n\nplt.imshow(tif_data)\nignore_tags = [\"StripOffsets\", \"StripByteCounts\"]\n# Print metadata\nfor page in tif.pages:\n    print(\"Page:\", page.index)\n    print(\"Shape:\", page.shape)\n    print(\"Tags:\")\n    for tag in page.tags.values():\n        if tag.name in ignore_tags:\n            continue\n            \n        print(tag.name, tag.value)\n    print(\"------------------\")\n\n# Close the TIF file\ntif.close()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:28:02.462645Z","iopub.execute_input":"2023-05-24T11:28:02.462981Z","iopub.status.idle":"2023-05-24T11:28:02.768606Z","shell.execute_reply.started":"2023-05-24T11:28:02.462951Z","shell.execute_reply":"2023-05-24T11:28:02.767581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(TRAIN_MASKS_FILE, 'r') as json_file:\n    json_labels = [json.loads(line) for line in json_file]\n    \nlabel_types = dict()\nfor json_label in json_labels:\n    annotations = json_label['annotations']\n\n    for item in annotations:\n        label_type = item[\"type\"]\n        if label_type not in label_types:\n            label_types[label_type] = 0\n        label_types[label_type] += 1\n\nprint(label_types)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T12:47:48.909650Z","iopub.execute_input":"2023-05-24T12:47:48.910097Z","iopub.status.idle":"2023-05-24T12:47:54.752679Z","shell.execute_reply.started":"2023-05-24T12:47:48.910062Z","shell.execute_reply":"2023-05-24T12:47:54.751162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_df = pd.read_csv(WSI_METADATA_FILE)\nprint(wsi_df.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-05-24T11:30:13.409082Z","iopub.execute_input":"2023-05-24T11:30:13.409409Z","iopub.status.idle":"2023-05-24T11:30:13.446436Z","shell.execute_reply.started":"2023-05-24T11:30:13.409387Z","shell.execute_reply":"2023-05-24T11:30:13.445698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Try to assemble whole slides\nThe larger datasets are missing random tiles. Haven't looked long enough to know if that is user error or by design or if we will see those missing tiles in test sets.","metadata":{"execution":{"iopub.status.busy":"2023-05-24T12:47:54.765789Z","iopub.execute_input":"2023-05-24T12:47:54.766275Z","iopub.status.idle":"2023-05-24T12:47:54.775293Z","shell.execute_reply.started":"2023-05-24T12:47:54.766232Z","shell.execute_reply":"2023-05-24T12:47:54.773649Z"}}},{"cell_type":"code","source":"\ndf = pd.read_csv(TILE_METADATA_FILE)\n\n# Using integer division directly instead of apply function\ndf[\"x\"] = df['i'] // 512\ndf[\"y\"] = df['j'] // 512\n\nfor dataset_id, df_dataset in df.groupby('dataset'):\n    unique_slides = df_dataset['source_wsi'].nunique()\n    \n    for slide_id, df_slide in df_dataset.groupby('source_wsi'):\n        tile_rows = df_slide['x']\n        tile_cols = df_slide['y']\n\n        x_start, x_end = tile_rows.min(), tile_rows.max() + 1\n        y_start, y_end = tile_cols.min(), tile_cols.max() + 1\n\n        num_rows, num_cols = x_end - x_start, y_end - y_start\n        size_x, size_y = num_rows * 512, num_cols * 512\n\n        slide_image =  np.full((size_y, size_x, 3), 255, np.uint8)  # white background\n        tiles_opened = 0\n\n        for xi, df_x in tqdm(df_slide.groupby('x')):\n            matched_rows = df_x\n            for yi, matched_tile in matched_rows.groupby('y'):\n                if not matched_tile.empty:\n                    x1 = (xi - x_start) * 512\n                    x2 = x1 + 512\n                    y1 = (yi - y_start) * 512\n                    y2 = y1 + 512\n                    tiles_opened += 1\n        \n                    source_file_name = f\"{matched_tile['id'].values[0]}.tif\"\n                    tile_path = os.path.join(TRAIN_IMAGE_DIR, source_file_name)\n\n                    tile_image = cv2.imread(tile_path, cv2.IMREAD_COLOR)\n                    tile_image = cv2.cvtColor(tile_image, cv2.COLOR_BGR2RGB)\n                    \n                    slide_image[y1:y2, x1:x2, :] = tile_image\n        total_slide_images = len(df_slide)\n        print(f\"dataset: {dataset_id}, slid_id: {slide_id}, total_images: {total_slide_images} images_opened: {tiles_opened}\")\n        fix, ax = plt.subplots(1,1, figsize=(32, 30))\n\n        d = ax.imshow(slide_image)\n        d = ax.set_title(f\"dataset: {dataset_id}, slid_id: {slide_id}\")\n        d = ax.grid(None)\n        d = ax.axis('off')\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T12:37:18.396662Z","iopub.execute_input":"2023-05-24T12:37:18.397097Z","iopub.status.idle":"2023-05-24T12:47:48.906614Z","shell.execute_reply.started":"2023-05-24T12:37:18.397064Z","shell.execute_reply":"2023-05-24T12:47:48.904550Z"},"trusted":true},"execution_count":null,"outputs":[]}]}