{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":75176,"databundleVersionId":8252256,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Explorative Data Analysis and Loading ","metadata":{"execution":{"iopub.status.busy":"2024-04-23T13:50:04.494867Z","iopub.execute_input":"2024-04-23T13:50:04.495349Z","iopub.status.idle":"2024-04-23T13:50:04.500000Z","shell.execute_reply.started":"2024-04-23T13:50:04.495319Z","shell.execute_reply":"2024-04-23T13:50:04.498974Z"}}},{"cell_type":"code","source":"# imports, helper functions and globals\nfrom pathlib import Path\nfrom typing import List, Tuple, Optional\n\nimport albumentations as A\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport torch\nfrom PIL import Image\nfrom torchvision.ops import nms\nfrom tqdm.notebook import tqdm, trange\n\n\ndef ls(path: Path) -> List[Path]:\n    return list(path.iterdir())\n\nROOT = Path(\"/kaggle/input/amia-public-challenge-2024\")\n\nCLASS_IDS_NAMES = {\n    0: 'Aortic enlargement',\n    1: 'Atelectasis',\n    2: 'Calcification',\n    3: 'Cardiomegaly',\n    4: 'Consolidation',\n    5: 'ILD',\n    6: 'Infiltration',\n    7: 'Lung Opacity',\n    8: 'Nodule/Mass',\n    9: 'Other lesion',\n    10: 'Pleural effusion',\n    11: 'Pleural thickening',\n    12: 'Pneumothorax',\n    13: 'Pulmonary fibrosis',\n    14: 'No finding'\n}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-24T14:20:20.371317Z","iopub.execute_input":"2024-04-24T14:20:20.371676Z","iopub.status.idle":"2024-04-24T14:20:25.429687Z","shell.execute_reply.started":"2024-04-24T14:20:20.371641Z","shell.execute_reply":"2024-04-24T14:20:25.428345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Folder structure","metadata":{}},{"cell_type":"code","source":"ls(ROOT)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:25.437490Z","iopub.execute_input":"2024-04-24T14:20:25.437816Z","iopub.status.idle":"2024-04-24T14:20:25.447047Z","shell.execute_reply.started":"2024-04-24T14:20:25.437788Z","shell.execute_reply":"2024-04-24T14:20:25.445445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of train image paths\nimages = ls(ROOT / \"train/train\")\nimages[:2]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:25.448829Z","iopub.execute_input":"2024-04-24T14:20:25.449279Z","iopub.status.idle":"2024-04-24T14:20:25.479901Z","shell.execute_reply.started":"2024-04-24T14:20:25.449235Z","shell.execute_reply":"2024-04-24T14:20:25.478673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of test image paths\ntest_images = ls(ROOT / \"train/train\")\ntest_images[:2]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:25.483324Z","iopub.execute_input":"2024-04-24T14:20:25.483694Z","iopub.status.idle":"2024-04-24T14:20:25.508175Z","shell.execute_reply.started":"2024-04-24T14:20:25.483662Z","shell.execute_reply":"2024-04-24T14:20:25.506951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore images","metadata":{}},{"cell_type":"code","source":"# Reading a single image\nImage.open(images[0]).resize((256, 256))","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:25.509428Z","iopub.execute_input":"2024-04-24T14:20:25.510554Z","iopub.status.idle":"2024-04-24T14:20:25.560149Z","shell.execute_reply.started":"2024-04-24T14:20:25.510505Z","shell.execute_reply":"2024-04-24T14:20:25.559046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some more examples\n\nnrows=3\nncols=10\n\nfig, axs = plt.subplots(nrows=nrows, ncols=ncols, sharey=True, sharex=True)\nfig.set_size_inches(ncols * 2, nrows * 2)\ni = 0\nfor row in range(nrows):\n    for col in range(ncols):\n        image = np.array(Image.open(images[i]))\n        axs[row, col].imshow(image, cmap=\"gray\")\n        i += 1\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:25.562028Z","iopub.execute_input":"2024-04-24T14:20:25.562413Z","iopub.status.idle":"2024-04-24T14:20:33.513367Z","shell.execute_reply.started":"2024-04-24T14:20:25.562383Z","shell.execute_reply":"2024-04-24T14:20:33.511982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets look at the pixel distributions\nfor i in range(20):\n    pixels = np.array(Image.open(images[i])).flatten()\n    count, _ = np.histogram(pixels, bins=256, range=(0, 255))\n    plt.plot(count, color=\"black\", alpha=0.5)\n    \nplt.ylabel(\"Count\")\n# plt.yscale(\"log\")\nplt.xlabel(\"Pixel value\")\nplt.grid(axis=\"y\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:33.514880Z","iopub.execute_input":"2024-04-24T14:20:33.515279Z","iopub.status.idle":"2024-04-24T14:20:34.489052Z","shell.execute_reply.started":"2024-04-24T14:20:33.515243Z","shell.execute_reply":"2024-04-24T14:20:34.487936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bounding Box Annotations\n\nAll images are rescaled to have shape (1024x1024). The bounding box coordinates still have to be adapted to this new image shape. image_size.csv contains the original image size and is used here to rescale the boudning box coordinates.  \nCheck explanations [here](https://www.kaggle.com/competitions/amia-public-challenge-2024/data).","metadata":{}},{"cell_type":"code","source":"def read_and_proccess_annotations(partition):\n    assert partition in [\"train\", \"test\"]\n    \n    # This dataframe contains the original image sizes.\n    # They are used to normalize the bounding box coordinates.\n    image_sizes = pd.read_csv(ROOT / \"img_size.csv\")\n    image_sizes.rename({\"dim0\": \"original_image_height\", \"dim1\": \"original_image_width\"},\n                       axis=1, inplace=True)\n    \n    # Read dataframe and merge with image_size information\n    df =  pd.read_csv(ROOT / f\"{partition}.csv\")\n    df = df.merge(image_sizes)\n    \n    # Normalize the bounding box coordinates to the resized images of shape (1024, 1024)\n    if partition == \"train\":\n        # Normalize coordinates accoring to height and width of the original images\n        df[\"x_min_norm\"] = (df.x_min / df.original_image_width) * 1024\n        df[\"x_max_norm\"] = (df.x_max / df.original_image_width) * 1024\n        df[\"y_min_norm\"] = (df.y_min / df.original_image_height) * 1024\n        df[\"y_max_norm\"] = (df.y_max / df.original_image_height) * 1024\n\n\n        # Compute the bounding box width and height\n        df[\"width\"] = df.x_max_norm - df.x_min_norm \n        df[\"height\"] = df.y_max_norm - df.y_min_norm\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:28:40.534333Z","iopub.execute_input":"2024-04-24T14:28:40.534827Z","iopub.status.idle":"2024-04-24T14:28:40.547322Z","shell.execute_reply.started":"2024-04-24T14:28:40.534788Z","shell.execute_reply":"2024-04-24T14:28:40.546151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = read_and_proccess_annotations(\"train\")\ntest_df = read_and_proccess_annotations(\"test\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:28:41.456733Z","iopub.execute_input":"2024-04-24T14:28:41.458130Z","iopub.status.idle":"2024-04-24T14:28:41.612499Z","shell.execute_reply.started":"2024-04-24T14:28:41.458080Z","shell.execute_reply":"2024-04-24T14:28:41.611268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:28:43.157115Z","iopub.execute_input":"2024-04-24T14:28:43.157530Z","iopub.status.idle":"2024-04-24T14:28:43.177754Z","shell.execute_reply.started":"2024-04-24T14:28:43.157494Z","shell.execute_reply":"2024-04-24T14:28:43.176809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:28:46.119893Z","iopub.execute_input":"2024-04-24T14:28:46.120305Z","iopub.status.idle":"2024-04-24T14:28:46.131519Z","shell.execute_reply.started":"2024-04-24T14:28:46.120272Z","shell.execute_reply":"2024-04-24T14:28:46.130022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Class Annotations","metadata":{}},{"cell_type":"code","source":"# For every sample we have multiple labels. This is a multi-label classification task!\n# Additionally we have labels from multiple radiologists (rad_id). \n# Every entry corresponds to a single bounding box.\n\ntrain_df[train_df.image_id == \"0FDQVdLgDKI1sRnPL94LzVh9EvXDVM9m\"]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:31:58.648967Z","iopub.execute_input":"2024-04-24T14:31:58.649404Z","iopub.status.idle":"2024-04-24T14:31:58.680154Z","shell.execute_reply.started":"2024-04-24T14:31:58.649372Z","shell.execute_reply":"2024-04-24T14:31:58.678959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have 15 unique classes!\nsorted(train_df.class_id.unique())","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:29:24.107214Z","iopub.execute_input":"2024-04-24T14:29:24.107647Z","iopub.status.idle":"2024-04-24T14:29:24.116270Z","shell.execute_reply.started":"2024-04-24T14:29:24.107615Z","shell.execute_reply":"2024-04-24T14:29:24.115360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mapping of class ids to class names\ntrain_df.groupby(\"class_id\").class_name.first().to_dict()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:29:24.137019Z","iopub.execute_input":"2024-04-24T14:29:24.138231Z","iopub.status.idle":"2024-04-24T14:29:24.151292Z","shell.execute_reply.started":"2024-04-24T14:29:24.138190Z","shell.execute_reply":"2024-04-24T14:29:24.150055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploration\n\nIdeas for classification labels:\n- Why are the more counts for some classes then we have samples?\n- Do we have class imbalance?\n- What is the frequency of every class?\n- Can we see classes accuring often together?\n\nIdeas for bounding boxes:\n- What are the size of the boxes for different classes?\n- Are the boxes equally distributed across the images? Any differences per class?\n- How do the outliers look?","metadata":{}},{"cell_type":"code","source":"counts = train_df['class_name'].value_counts()\n\nsns.barplot(y=counts.index, x=counts.values, color=\"gray\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:34.755169Z","iopub.execute_input":"2024-04-24T14:20:34.755515Z","iopub.status.idle":"2024-04-24T14:20:35.075433Z","shell.execute_reply.started":"2024-04-24T14:20:34.755487Z","shell.execute_reply":"2024-04-24T14:20:35.074508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.FacetGrid(train_df, col=\"class_name\", col_wrap=5)\ng.map(sns.scatterplot, \"width\", \"height\", alpha=0.2);","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:35.076632Z","iopub.execute_input":"2024-04-24T14:20:35.077230Z","iopub.status.idle":"2024-04-24T14:20:40.225518Z","shell.execute_reply.started":"2024-04-24T14:20:35.077198Z","shell.execute_reply":"2024-04-24T14:20:40.224375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset class","metadata":{}},{"cell_type":"code","source":"def to_one_hot_encoded(class_indeces, num_classes=15) -> torch.Tensor:\n    one_hot_encoded = torch.zeros(num_classes)\n    one_hot_encoded[torch.unique(torch.tensor(class_indeces))] = 1.\n    return one_hot_encoded\n\ndef plot_bounding_boxes(image, bounding_boxes, class_ids=None, ax=None):\n    \n    if ax is None:\n        plt.imshow(image, cmap=\"gray\")\n        ax = plt.gca()\n    else:\n        ax.imshow(image, cmap=\"gray\")\n    \n    if class_ids is not None:\n        colors = [mpl.colormaps[\"tab20b\"](i) for i in class_ids]\n    \n    for i, bbox in enumerate(bounding_boxes):\n        xmin, ymin, xmax, ymax = bbox\n        width = xmax - xmin\n        height = ymax - ymin\n        rect = plt.Rectangle((xmin, ymin), width, height, fill=False, edgecolor=colors[i], linewidth=2)\n        ax.add_patch(rect)\n        if class_ids is not None:\n            class_id = class_ids[i]\n            ax.text(xmin, ymin - 10, f'{CLASS_IDS_NAMES[class_id]}', bbox=dict(facecolor=colors[i], alpha=0.2), fontsize=12, color='white')\n    \n    return ax","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:22:18.281987Z","iopub.execute_input":"2024-04-24T14:22:18.282754Z","iopub.status.idle":"2024-04-24T14:22:18.294333Z","shell.execute_reply.started":"2024-04-24T14:22:18.282716Z","shell.execute_reply":"2024-04-24T14:22:18.292934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AMIADataset:\n    def __init__(self, partition=\"train\", iou_threshold=0.7, transform=None):\n          \n        self.num_classes = 15        \n        self.iou_threshold = iou_threshold\n\n        self.data = read_and_proccess_annotations(partition)\n        self.images = ls(ROOT / f\"{partition}/{partition}\")\n        \n        self.transform = transform\n    \n    def __len__(self) -> int:\n        return len(self.images)\n    \n    def _read_bounding_box(self, sample: pd.Series) -> Tuple[torch.Tensor, List]:\n        \"\"\"Process bounding boxes for a given sample.\"\"\"\n\n        bboxes = sample.loc[:, [\"x_min_norm\", \"y_min_norm\", \"x_max_norm\", \"y_max_norm\"]]\n        bboxes = torch.tensor(bboxes.values).float() / 1024 # Normalize to 0 and 1.\n        \n        # Apply Non-maximum Supression to remove highly overlapping bounding boxes.\n        boxes_to_keep = nms(bboxes, torch.ones(len(bboxes)), self.iou_threshold)\n        bboxes = bboxes[boxes_to_keep]\n        \n        return bboxes, boxes_to_keep.tolist()\n    \n    def __getitem__(self, idx: int) -> Tuple[np.ndarray, List[int], Optional[torch.Tensor]]:\n        \n        # Get sample by image_id\n        image_path = self.images[idx]\n        image_id = image_path.stem\n        image = np.array(Image.open(image_path))\n\n        sample = self.data[self.data.image_id == image_id]\n\n        # Generate one hot encode target vector\n        labels = to_one_hot_encoded(sample.class_id.unique())\n\n        # Generate bounding box list           \n        bboxes, boxes_to_keep = self._read_bounding_box(sample)\n\n        # Only keep class labels of kept bounding bobxes\n        class_labels = sample.class_id.iloc[boxes_to_keep].tolist()\n\n        has_bbox = torch.isnan(bboxes).sum() == 0\n\n        # Generate fake bbox for transformation\n        if not has_bbox:\n            bboxes = torch.zeros(bboxes.shape[0], 4)\n            bboxes[:, 2:] += 0.1\n\n        if self.transform is not None:\n            transformed = transform(image=image, bboxes=bboxes, class_labels=class_labels)\n            image = transformed['image']\n            bboxes = torch.tensor(transformed['bboxes'])\n\n        # Remove fake bbox\n        if not has_bbox:\n            bboxes = None\n\n        return image, class_labels, bboxes","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:22:19.144142Z","iopub.execute_input":"2024-04-24T14:22:19.144567Z","iopub.status.idle":"2024-04-24T14:22:19.159085Z","shell.execute_reply.started":"2024-04-24T14:22:19.144532Z","shell.execute_reply":"2024-04-24T14:22:19.158192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = AMIADataset(iou_threshold=0.7)\nimage, class_ids, bboxes = dataset[1]\n\nplot_bounding_boxes(image, bboxes * 1024, class_ids);","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:22:19.933066Z","iopub.execute_input":"2024-04-24T14:22:19.934081Z","iopub.status.idle":"2024-04-24T14:22:20.535221Z","shell.execute_reply.started":"2024-04-24T14:22:19.934037Z","shell.execute_reply":"2024-04-24T14:22:20.534358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_one_hot_encoded(class_ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:22:20.974888Z","iopub.execute_input":"2024-04-24T14:22:20.975862Z","iopub.status.idle":"2024-04-24T14:22:20.985509Z","shell.execute_reply.started":"2024-04-24T14:22:20.975811Z","shell.execute_reply":"2024-04-24T14:22:20.984234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Adding augmentations using *albumentation*.\nFor more info check:\n- [Overview of augmentations](https://albumentations.ai/docs/getting_started/transforms_and_targets/)\n- [Bounding box tutorial](https://albumentations.ai/docs/getting_started/bounding_boxes_augmentation/)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T21:29:09.047218Z","iopub.execute_input":"2024-04-23T21:29:09.047790Z","iopub.status.idle":"2024-04-23T21:29:09.056893Z","shell.execute_reply.started":"2024-04-23T21:29:09.047754Z","shell.execute_reply":"2024-04-23T21:29:09.054798Z"}}},{"cell_type":"code","source":"transform = A.Compose([\n    A.RandomCrop(width=900, height=900),\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(p=0.2),\n], bbox_params=A.BboxParams(format='albumentations', label_fields=['class_labels'], min_visibility=0.2))\n\ndataset = AMIADataset(transform=transform, iou_threshold=0.7)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:41.149547Z","iopub.execute_input":"2024-04-24T14:20:41.149893Z","iopub.status.idle":"2024-04-24T14:20:41.286947Z","shell.execute_reply.started":"2024-04-24T14:20:41.149840Z","shell.execute_reply":"2024-04-24T14:20:41.285541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=3, sharex=True, sharey=True)\n\nfig.set_size_inches(12, 4)\n\nfor i in range(3):\n    image, class_ids, bboxes = dataset[1]\n    plot_bounding_boxes(image, bboxes * 1024, class_ids, ax=axs[i])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:41.289036Z","iopub.execute_input":"2024-04-24T14:20:41.289983Z","iopub.status.idle":"2024-04-24T14:20:42.406703Z","shell.execute_reply.started":"2024-04-24T14:20:41.289938Z","shell.execute_reply":"2024-04-24T14:20:42.405714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image statistics","metadata":{}},{"cell_type":"code","source":"dataset = AMIADataset()\n\n# Quick way to find dataset mean. Only sample a few images.\nimages_to_sample = 1000\nimages = np.stack([dataset[i][0] for i in trange(images_to_sample)])","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:20:42.408155Z","iopub.execute_input":"2024-04-24T14:20:42.408756Z","iopub.status.idle":"2024-04-24T14:21:10.195325Z","shell.execute_reply.started":"2024-04-24T14:20:42.408716Z","shell.execute_reply":"2024-04-24T14:21:10.194159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = np.mean(images[images > 0])\nstd = np.std(images[images > 0])\n\nprint(f\"Image mean \\t= {mean:.2f} \\nImage std \\t= {std:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-24T14:21:10.197403Z","iopub.execute_input":"2024-04-24T14:21:10.197892Z","iopub.status.idle":"2024-04-24T14:21:19.345015Z","shell.execute_reply.started":"2024-04-24T14:21:10.197833Z","shell.execute_reply":"2024-04-24T14:21:19.343752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}