{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\nimporting the necessary libraries\n\"\"\"\n\nimport pandas as pd\nimport os\nimport sys\nimport json\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom PIL import Image\n\nimport tensorflow as tf\nimport albumentations as A\nimport numpy as np\nimport cv2\nimport tifffile as tiff","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-13T03:22:41.154318Z","iopub.execute_input":"2023-06-13T03:22:41.154694Z","iopub.status.idle":"2023-06-13T03:22:52.200936Z","shell.execute_reply.started":"2023-06-13T03:22:41.154662Z","shell.execute_reply":"2023-06-13T03:22:52.200003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nTHe below pieces of code is directly from one of the submitters who was kind enough to share\n\"\"\"\n\nclass Acquisition:\n    \n    def get_datframe(self,path):\n        return pd.read_csv(path)\n    \n    def get_json_dataframe(self, json_file):\n        data = []\n        with open(json_file, 'r') as file:\n            for line in file:\n                item = json.loads(line)\n                data.append(item)\n        \n        json_df = pd.DataFrame(data)\n        return json_df\n    \n        \n        \nacq = Acquisition()    \n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:22:52.203092Z","iopub.execute_input":"2023-06-13T03:22:52.203580Z","iopub.status.idle":"2023-06-13T03:22:52.211468Z","shell.execute_reply.started":"2023-06-13T03:22:52.203542Z","shell.execute_reply":"2023-06-13T03:22:52.209764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"title=acq.get_datframe('/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv')\ntitle.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:22:52.213101Z","iopub.execute_input":"2023-06-13T03:22:52.213750Z","iopub.status.idle":"2023-06-13T03:22:52.284755Z","shell.execute_reply.started":"2023-06-13T03:22:52.213713Z","shell.execute_reply":"2023-06-13T03:22:52.282617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi = acq.get_datframe(path='/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv')\nwsi.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:22:52.289258Z","iopub.execute_input":"2023-06-13T03:22:52.289855Z","iopub.status.idle":"2023-06-13T03:22:52.323768Z","shell.execute_reply.started":"2023-06-13T03:22:52.289758Z","shell.execute_reply":"2023-06-13T03:22:52.321634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"polygons_df = acq.get_json_dataframe('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl')\npolygons_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:22:52.325994Z","iopub.execute_input":"2023-06-13T03:22:52.326898Z","iopub.status.idle":"2023-06-13T03:22:56.960846Z","shell.execute_reply.started":"2023-06-13T03:22:52.326857Z","shell.execute_reply":"2023-06-13T03:22:56.959802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nWe are separating the annotations from the json file to the very atomic state.\nEach annotation is a new row in the created dataframe\n\"\"\"\n\ndef separate_annotations(dataset):\n    separated  = pd.DataFrame(columns=[\"id\",\"type\",\"coordinates\"])\n    for index in dataset.index:\n        id = dataset[\"id\"][index]\n        all_annotations = dataset[\"annotations\"][index]\n        for each_annotation in all_annotations:\n            annotation_type = each_annotation[\"type\"]\n            annotation_coordinates = each_annotation[\"coordinates\"]\n            separated.loc[len(separated)]=[id,annotation_type,annotation_coordinates]\n    return separated\n\nseparated_polygons_df  = separate_annotations(polygons_df)\nseparated_polygons_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:22:56.962170Z","iopub.execute_input":"2023-06-13T03:22:56.962755Z","iopub.status.idle":"2023-06-13T03:23:37.414898Z","shell.execute_reply.started":"2023-06-13T03:22:56.962716Z","shell.execute_reply":"2023-06-13T03:23:37.413766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"The below code for augmentation seems to really good and has been copied from another kaggle user's notebook\"\"\"\n\n\ndef get_augmentation(p=1.0):\n    return A.Compose([\n        A.HorizontalFlip(),\n        A.VerticalFlip(),\n        A.RandomRotate90(),\n        A.ShiftScaleRotate(shift_limit=0.0625, scale_limit=0.2, rotate_limit=15, p=0.9,\n                         border_mode=cv2.BORDER_REFLECT),\n        A.OneOf([\n            A.ElasticTransform(p=.3),\n            A.GaussianBlur(p=.3),\n            A.GaussNoise(p=.3),\n            A.OpticalDistortion(p=0.3),\n            A.GridDistortion(p=.1),\n        ], p=0.3),\n        A.OneOf([\n            A.HueSaturationValue(15,25,0),\n            A.CLAHE(clip_limit=2),\n            A.RandomBrightnessContrast(brightness_limit=0.3, contrast_limit=0.3),\n        ], p=0.3),\n    ], p=p)","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:23:37.416179Z","iopub.execute_input":"2023-06-13T03:23:37.416563Z","iopub.status.idle":"2023-06-13T03:23:37.423692Z","shell.execute_reply.started":"2023-06-13T03:23:37.416537Z","shell.execute_reply":"2023-06-13T03:23:37.422677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nSince tensorflow does not readily support tiff files, we use a library called tifffile and the convert\nthe images to tensorflow file format.\n\"\"\"\n\ndef preprocess_image(file_name, image_size=(512, 512), augmentation=None):\n    path = '/kaggle/input/hubmap-hacking-the-human-vasculature/train/{}.tif'.format(file_name)\n    image = tiff.imread(path)\n    image = tf.convert_to_tensor(image, dtype=tf.uint8)\n    image = tf.image.resize(image, image_size)\n    image = np.asarray(image, dtype = np.uint8)\n    plt.imshow(image)\n    \n    mask = np.zeros(image_size, dtype=np.float32)\n    filter_criteria = separated_polygons_df[\"id\"] == file_name\n    all_coordinates = separated_polygons_df.loc[filter_criteria, \"coordinates\"].tolist()\n    all_type = separated_polygons_df.loc[filter_criteria, \"type\"].tolist()\n    for i in range(len(all_coordinates)):\n        if all_type[i] == \"blood_vessel\":\n            x_values = [point[0] for point in all_coordinates[i][0]]\n            y_values = [point[1] for point in all_coordinates[i][0]]\n            mask[x_values, y_values] = 1\n    # Apply data augmentation if provided\n    if augmentation is not None:\n        augmented = augmentation(image=image, mask=mask)\n        image, mask = augmented[\"image\"]/255.0,augmented[\"mask\"]\n\n    return image, mask\n\n\nsample_image = separated_polygons_df[\"id\"][0]\nimage, mask = preprocess_image(sample_image, augmentation=get_augmentation(p=1.0))\n\n# Display the image and mask\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\n\nif image is not None or mask is not None:\n    axes[0].imshow(image)\n    axes[0].set_title(\"Image\")\n    axes[0].axis(\"off\")\n    axes[1].imshow(mask)\n    axes[1].set_title(\"Mask\")\n    axes[1].axis(\"off\")\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:23:37.425182Z","iopub.execute_input":"2023-06-13T03:23:37.426062Z","iopub.status.idle":"2023-06-13T03:23:38.315876Z","shell.execute_reply.started":"2023-06-13T03:23:37.426032Z","shell.execute_reply":"2023-06-13T03:23:38.314845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_unique_images = separated_polygons_df[\"id\"].unique()\n\nimages=[]\nmasks=[]\nfor each_image in all_unique_images:\n    image, mask = preprocess_image(each_image, augmentation=get_augmentation(p=1.0))\n    images.append(image)\n    masks.append(mask)\n  \nimages = tf.convert_to_tensor(images)\nmasks = tf.convert_to_tensor(masks)\n\n# Create a TensorFlow Dataset\ndataset = tf.data.Dataset.from_tensor_slices((images, masks))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T03:23:38.317267Z","iopub.execute_input":"2023-06-13T03:23:38.317669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset.shuffle(buffer_size=len(images))\n# Define the split percentages\ntrain_split = 0.7\nval_split = 0.15\ntest_split = 0.15\nbatch_size = 1\n\n# Calculate the sizes of each split\nnum_samples = len(images)\nnum_train_samples = int(train_split * num_samples)\nnum_val_samples = int(val_split * num_samples)\nnum_test_samples = num_samples - num_train_samples - num_val_samples\n\n# Split the dataset\ntrain_dataset = dataset.take(num_train_samples).batch(batch_size)\nval_dataset = dataset.skip(num_train_samples).take(num_val_samples).batch(batch_size)\ntest_dataset = dataset.skip(num_train_samples + num_val_samples).take(num_test_samples).batch(batch_size)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}