{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":75176,"databundleVersionId":8252256,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Explorative Data Analysis and Loading ","metadata":{"execution":{"iopub.status.busy":"2024-04-23T13:50:04.494867Z","iopub.execute_input":"2024-04-23T13:50:04.495349Z","iopub.status.idle":"2024-04-23T13:50:04.5Z","shell.execute_reply.started":"2024-04-23T13:50:04.495319Z","shell.execute_reply":"2024-04-23T13:50:04.498974Z"}}},{"cell_type":"code","source":"!pip install iterative-stratification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:12.593340Z","iopub.execute_input":"2026-09-04T14:46:12.594302Z","iopub.status.idle":"2026-09-04T14:46:23.935871Z","shell.execute_reply.started":"2026-09-04T14:46:12.594264Z","shell.execute_reply":"2026-09-04T14:46:23.934706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# imports, helper functions and globals\nfrom pathlib import Path\nfrom typing import List, Tuple, Optional\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport torch\nimport gc\nfrom PIL import Image\nimport torch.nn as nn\nimport torchvision.models as models\nimport torchvision.transforms as T\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision.ops import nms\nfrom tqdm.notebook import tqdm, trange\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold\nfrom sklearn.metrics import (\n    roc_auc_score,\n    average_precision_score,\n    f1_score,\n    precision_score,\n    recall_score,\n)\n\n\ndef ls(path: Path) -> List[Path]:\n    return list(path.iterdir())\n\nROOT = Path(\"/kaggle/input/amia-public-challenge-2024\")\n\nCLASS_IDS_NAMES = {\n    0: 'Aortic enlargement',\n    1: 'Atelectasis',\n    2: 'Calcification',\n    3: 'Cardiomegaly',\n    4: 'Consolidation',\n    5: 'ILD',\n    6: 'Infiltration',\n    7: 'Lung Opacity',\n    8: 'Nodule/Mass',\n    9: 'Other lesion',\n    10: 'Pleural effusion',\n    11: 'Pleural thickening',\n    12: 'Pneumothorax',\n    13: 'Pulmonary fibrosis',\n    14: 'No finding'\n}\nNUM_CLASSES = 15","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:23.938133Z","iopub.execute_input":"2026-09-04T14:46:23.938953Z","iopub.status.idle":"2026-09-04T14:46:35.616300Z","shell.execute_reply.started":"2026-09-04T14:46:23.938915Z","shell.execute_reply":"2026-09-04T14:46:35.615589Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Folder structure","metadata":{}},{"cell_type":"code","source":"ls(ROOT)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:35.617391Z","iopub.execute_input":"2026-09-04T14:46:35.617798Z","iopub.status.idle":"2026-09-04T14:46:35.625237Z","shell.execute_reply.started":"2026-09-04T14:46:35.617775Z","shell.execute_reply":"2026-09-04T14:46:35.624589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of train image paths\nimages = ls(ROOT / \"train/train\")\nimages[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:35.627695Z","iopub.execute_input":"2026-09-04T14:46:35.628159Z","iopub.status.idle":"2026-09-04T14:46:35.723497Z","shell.execute_reply.started":"2026-09-04T14:46:35.628138Z","shell.execute_reply":"2026-09-04T14:46:35.722620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of test image paths\ntest_images = ls(ROOT / \"test/test\")\ntest_images[:2]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:35.724597Z","iopub.execute_input":"2026-09-04T14:46:35.724939Z","iopub.status.idle":"2026-09-04T14:46:35.877429Z","shell.execute_reply.started":"2026-09-04T14:46:35.724909Z","shell.execute_reply":"2026-09-04T14:46:35.876459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Explore images","metadata":{}},{"cell_type":"code","source":"# Reading a single image\nImage.open(images[0]).resize((256, 256))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:35.878704Z","iopub.execute_input":"2026-09-04T14:46:35.879076Z","iopub.status.idle":"2026-09-04T14:46:35.942283Z","shell.execute_reply.started":"2026-09-04T14:46:35.879045Z","shell.execute_reply":"2026-09-04T14:46:35.941321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Some more examples\n\nnrows=3\nncols=10\n\nfig, axs = plt.subplots(nrows=nrows, ncols=ncols, sharey=True, sharex=True)\nfig.set_size_inches(ncols * 2, nrows * 2)\ni = 0\nfor row in range(nrows):\n    for col in range(ncols):\n        image = np.array(Image.open(images[i]))\n        axs[row, col].imshow(image, cmap=\"gray\")\n        i += 1\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:35.943248Z","iopub.execute_input":"2026-09-04T14:46:35.943504Z","iopub.status.idle":"2026-09-04T14:46:43.067330Z","shell.execute_reply.started":"2026-09-04T14:46:35.943485Z","shell.execute_reply":"2026-09-04T14:46:43.066477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lets look at the pixel distributions\nfor i in range(20):\n    pixels = np.array(Image.open(images[i])).flatten()\n    count, _ = np.histogram(pixels, bins=256, range=(0, 255))\n    plt.plot(count, color=\"black\", alpha=0.5)\n    \nplt.ylabel(\"Count\")\n# plt.yscale(\"log\")\nplt.xlabel(\"Pixel value\")\nplt.grid(axis=\"y\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:43.068682Z","iopub.execute_input":"2026-09-04T14:46:43.069001Z","iopub.status.idle":"2026-09-04T14:46:43.901737Z","shell.execute_reply.started":"2026-09-04T14:46:43.068974Z","shell.execute_reply":"2026-09-04T14:46:43.900916Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bounding Box Annotations\n\nAll images are rescaled to have shape (1024x1024). The bounding box coordinates still have to be adapted to this new image shape. image_size.csv contains the original image size and is used here to rescale the boudning box coordinates.  \nCheck explanations [here](https://www.kaggle.com/competitions/amia-public-challenge-2024/data).","metadata":{}},{"cell_type":"code","source":"def read_and_proccess_annotations(partition):\n    assert partition in [\"train\", \"test\"]\n    \n    # This dataframe contains the original image sizes.\n    # They are used to normalize the bounding box coordinates.\n    image_sizes = pd.read_csv(ROOT / \"img_size.csv\")\n    image_sizes.rename({\"dim0\": \"original_image_height\", \"dim1\": \"original_image_width\"},\n                       axis=1, inplace=True)\n    \n    # Read dataframe and merge with image_size information\n    df =  pd.read_csv(ROOT / f\"{partition}.csv\")\n    df = df.merge(image_sizes)\n    \n    # Normalize the bounding box coordinates to the resized images of shape (1024, 1024)\n    if partition == \"train\":\n        # Normalize coordinates accoring to height and width of the original images\n        df[\"x_min_norm\"] = (df.x_min / df.original_image_width) * 1024\n        df[\"x_max_norm\"] = (df.x_max / df.original_image_width) * 1024\n        df[\"y_min_norm\"] = (df.y_min / df.original_image_height) * 1024\n        df[\"y_max_norm\"] = (df.y_max / df.original_image_height) * 1024\n\n\n        # Compute the bounding box width and height\n        df[\"width\"] = df.x_max_norm - df.x_min_norm \n        df[\"height\"] = df.y_max_norm - df.y_min_norm\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:43.903057Z","iopub.execute_input":"2026-09-04T14:46:43.903385Z","iopub.status.idle":"2026-09-04T14:46:43.910732Z","shell.execute_reply.started":"2026-09-04T14:46:43.903356Z","shell.execute_reply":"2026-09-04T14:46:43.909791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = read_and_proccess_annotations(\"train\")\ntest_df = read_and_proccess_annotations(\"test\")\nprint(train_df.shape)\nprint(test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:43.913415Z","iopub.execute_input":"2026-09-04T14:46:43.913737Z","iopub.status.idle":"2026-09-04T14:46:44.173786Z","shell.execute_reply.started":"2026-09-04T14:46:43.913715Z","shell.execute_reply":"2026-09-04T14:46:44.172495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.175058Z","iopub.execute_input":"2026-09-04T14:46:44.175401Z","iopub.status.idle":"2026-09-04T14:46:44.199469Z","shell.execute_reply.started":"2026-09-04T14:46:44.175370Z","shell.execute_reply":"2026-09-04T14:46:44.198419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.200665Z","iopub.execute_input":"2026-09-04T14:46:44.201036Z","iopub.status.idle":"2026-09-04T14:46:44.208848Z","shell.execute_reply.started":"2026-09-04T14:46:44.201004Z","shell.execute_reply":"2026-09-04T14:46:44.207976Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class Annotations","metadata":{}},{"cell_type":"code","source":"# For every sample we have multiple labels. This is a multi-label classification task!\n# Additionally we have labels from multiple radiologists (rad_id). \n# Every entry corresponds to a single bounding box.\n\ntrain_df[train_df.image_id == \"0FDQVdLgDKI1sRnPL94LzVh9EvXDVM9m\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.210096Z","iopub.execute_input":"2026-09-04T14:46:44.210445Z","iopub.status.idle":"2026-09-04T14:46:44.236537Z","shell.execute_reply.started":"2026-09-04T14:46:44.210412Z","shell.execute_reply":"2026-09-04T14:46:44.235527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We have 15 unique classes!\nsorted(train_df.class_id.unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.237758Z","iopub.execute_input":"2026-09-04T14:46:44.238093Z","iopub.status.idle":"2026-09-04T14:46:44.253105Z","shell.execute_reply.started":"2026-09-04T14:46:44.238060Z","shell.execute_reply":"2026-09-04T14:46:44.252184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping of class ids to class names\ntrain_df.groupby(\"class_id\").class_name.first().to_dict()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.254005Z","iopub.execute_input":"2026-09-04T14:46:44.254273Z","iopub.status.idle":"2026-09-04T14:46:44.276843Z","shell.execute_reply.started":"2026-09-04T14:46:44.254254Z","shell.execute_reply":"2026-09-04T14:46:44.275644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Exploration\n\nIdeas for classification labels:\n- Why are the more counts for some classes then we have samples?\n- Do we have class imbalance?\n- What is the frequency of every class?\n- Can we see classes accuring often together?\n\nIdeas for bounding boxes:\n- What are the size of the boxes for different classes?\n- Are the boxes equally distributed across the images? Any differences per class?\n- How do the outliers look?","metadata":{}},{"cell_type":"code","source":"counts = train_df['class_name'].value_counts()\n\nsns.barplot(y=counts.index, x=counts.values, color=\"gray\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.278056Z","iopub.execute_input":"2026-09-04T14:46:44.278405Z","iopub.status.idle":"2026-09-04T14:46:44.513611Z","shell.execute_reply.started":"2026-09-04T14:46:44.278365Z","shell.execute_reply":"2026-09-04T14:46:44.512827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Long-tail Analysis - Analyse the frequency of every class\n\ncounts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.514904Z","iopub.execute_input":"2026-09-04T14:46:44.515244Z","iopub.status.idle":"2026-09-04T14:46:44.522703Z","shell.execute_reply.started":"2026-09-04T14:46:44.515213Z","shell.execute_reply":"2026-09-04T14:46:44.521769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Multi-label Analysis - Analyse patients with no, 1 or multiple abnormalities \n\nmulti_class_count = train_df['image_id'].value_counts()\nmulti_class_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.523918Z","iopub.execute_input":"2026-09-04T14:46:44.524198Z","iopub.status.idle":"2026-09-04T14:46:44.545435Z","shell.execute_reply.started":"2026-09-04T14:46:44.524179Z","shell.execute_reply":"2026-09-04T14:46:44.544591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Co-occurrence matrix - What abnormalities usually occur together?\n\ngroups = train_df.groupby(\"image_id\")[\"class_id\"].unique()\n\nco_occurrence = np.zeros((15, 15), dtype=int)\n\nfor classes in groups:\n    for i in classes:\n        for j in classes:\n            co_occurrence[i, j] += 1\nco_occurrence","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:44.546458Z","iopub.execute_input":"2026-09-04T14:46:44.546807Z","iopub.status.idle":"2026-09-04T14:46:45.014824Z","shell.execute_reply.started":"2026-09-04T14:46:44.546786Z","shell.execute_reply":"2026-09-04T14:46:45.013855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g = sns.FacetGrid(train_df, col=\"class_name\", col_wrap=5)\ng.map(sns.scatterplot, \"width\", \"height\", alpha=0.2);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:45.016206Z","iopub.execute_input":"2026-09-04T14:46:45.016654Z","iopub.status.idle":"2026-09-04T14:46:49.040671Z","shell.execute_reply.started":"2026-09-04T14:46:45.016615Z","shell.execute_reply":"2026-09-04T14:46:49.039646Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset class","metadata":{}},{"cell_type":"code","source":"def to_one_hot_encoded(class_indeces, num_classes=15) -> torch.Tensor:\n    one_hot_encoded = torch.zeros(num_classes)\n    one_hot_encoded[torch.unique(torch.tensor(class_indeces))] = 1.\n    return one_hot_encoded\n\ndef plot_bounding_boxes(image, bounding_boxes, class_ids=None, ax=None):\n    img = image.permute(1, 2, 0).cpu().numpy()\n    if ax is None:\n        plt.imshow(img, cmap=\"gray\")\n        ax = plt.gca()\n    else:\n        ax.imshow(img, cmap=\"gray\")\n    \n    if class_ids is not None:\n        colors = [mpl.colormaps[\"tab20b\"](i) for i in class_ids]\n    \n    for i, bbox in enumerate(bounding_boxes):\n        xmin, ymin, xmax, ymax = bbox\n        width = xmax - xmin\n        height = ymax - ymin\n        rect = plt.Rectangle((xmin, ymin), width, height, fill=False, edgecolor=colors[i], linewidth=2)\n        ax.add_patch(rect)\n        if class_ids is not None:\n            class_id = class_ids[i]\n            ax.text(xmin, ymin - 10, f'{CLASS_IDS_NAMES[class_id]}', bbox=dict(facecolor=colors[i], alpha=0.2), fontsize=12, color='white')\n    \n    return ax","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:49.041782Z","iopub.execute_input":"2026-09-04T14:46:49.042125Z","iopub.status.idle":"2026-09-04T14:46:49.050725Z","shell.execute_reply.started":"2026-09-04T14:46:49.042102Z","shell.execute_reply":"2026-09-04T14:46:49.049779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AMIADataset(Dataset):\n    def __init__(self, partition=\"train\", image_ids=None, labels_df=None, iou_threshold=0.7, transform=None):\n        self.partition = partition\n        self.num_classes = NUM_CLASSES\n        self.iou_threshold = iou_threshold\n\n        # Transformation \n        if transform is None:\n            self.transform = A.Compose(\n                [\n                    A.Resize(512, 512),\n            \n                    A.Normalize(\n                        mean=[0.485, 0.485, 0.485],\n                        std=[0.229, 0.229, 0.229],\n                    ),\n            \n                    ToTensorV2(),\n                ],\n                bbox_params=A.BboxParams(format=\"albumentations\", label_fields=[\"class_labels\"]),\n            )\n        else:\n            self.transform = transform\n\n        # Annotation dataframe\n        self.data = read_and_proccess_annotations(partition)\n\n        # Image paths\n        self.images = sorted(ls(ROOT / f\"{partition}/{partition}\"))\n\n        # Map image_id -> path\n        self.image_map = {\n            path.stem: path\n            for path in self.images\n        }\n\n        # Image IDs used by this dataset\n        if image_ids is None:\n            self.image_ids = list(self.image_map.keys())\n        else:\n            self.image_ids = list(image_ids)\n\n        # Labels\n        if labels_df is not None:\n            self.labels_df = labels_df.reindex(self.image_ids).fillna(0).astype(np.int32)\n        else:\n            self.labels_df = None\n\n    def __len__(self):\n        return len(self.image_ids)\n\n    def _get_bboxes(self, image_id):\n\n        sample = self.data[self.data.image_id == image_id]\n\n        # Test set has no bounding boxes\n        if len(sample) == 0 or \"x_min_norm\" not in sample.columns:\n            return (\n                torch.empty((0, 4), dtype=torch.float32),\n                []\n            )\n\n        bbox_columns = [\n            \"x_min_norm\",\n            \"y_min_norm\",\n            \"x_max_norm\",\n            \"y_max_norm\",\n        ]\n\n        bboxes = (\n            sample[bbox_columns]\n            .to_numpy(dtype=np.float32)\n            / 1024.0\n        )\n\n        bboxes = torch.tensor(bboxes)\n\n        # Remove invalid boxes\n        valid = ~torch.isnan(bboxes).any(dim=1)\n        bboxes = bboxes[valid]\n\n        class_ids = (\n            sample.loc[valid.numpy(), \"class_id\"]\n            .astype(int)\n            .tolist()\n        )\n\n        if len(bboxes) == 0:\n            return (\n                torch.empty((0, 4), dtype=torch.float32),\n                []\n            )\n\n        # NMS\n        keep = nms(\n            bboxes,\n            torch.ones(len(bboxes)),\n            self.iou_threshold\n        )\n\n        bboxes = bboxes[keep]\n        class_ids = [class_ids[i] for i in keep.tolist()]\n\n        return bboxes, class_ids\n\n    def __getitem__(self, idx):\n\n        image_id = self.image_ids[idx]\n        image_path = self.image_map[image_id]\n\n        image = np.array(\n            Image.open(image_path).convert(\"RGB\")\n        )\n\n        # ----------------------------------------------------\n        # Target\n        # ----------------------------------------------------\n\n        if self.labels_df is not None:\n            target = torch.tensor(\n                self.labels_df.loc[image_id].values,\n                dtype=torch.int32\n            )\n        else:\n            target = torch.zeros(\n                self.num_classes,\n                dtype=torch.int32\n            )\n\n        # ----------------------------------------------------\n        # Bounding boxes\n        # ----------------------------------------------------\n\n        bboxes, class_ids = self._get_bboxes(image_id)\n\n        # Albumentations expects list\n        bbox_list = bboxes.tolist()\n\n        # ----------------------------------------------------\n        # Transform\n        # ----------------------------------------------------\n\n        if self.transform is not None:\n\n            transformed = self.transform(\n                image=image,\n                bboxes=bbox_list,\n                class_labels=class_ids,\n            )\n\n            image = transformed[\"image\"]\n            bboxes = torch.tensor(\n                transformed[\"bboxes\"],\n                dtype=torch.float32\n            )\n\n        return image, target, bboxes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:49.052095Z","iopub.execute_input":"2026-09-04T14:46:49.052360Z","iopub.status.idle":"2026-09-04T14:46:49.072896Z","shell.execute_reply.started":"2026-09-04T14:46:49.052341Z","shell.execute_reply":"2026-09-04T14:46:49.072161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = AMIADataset(iou_threshold=0.7)\nimage, class_ids, bboxes = dataset[1]\n\nplot_bounding_boxes(image, bboxes * 1024, class_ids);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:49.073832Z","iopub.execute_input":"2026-09-04T14:46:49.074144Z","iopub.status.idle":"2026-09-04T14:46:49.984472Z","shell.execute_reply.started":"2026-09-04T14:46:49.074113Z","shell.execute_reply":"2026-09-04T14:46:49.983637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"to_one_hot_encoded(class_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:49.985653Z","iopub.execute_input":"2026-09-04T14:46:49.986225Z","iopub.status.idle":"2026-09-04T14:46:50.045321Z","shell.execute_reply.started":"2026-09-04T14:46:49.986201Z","shell.execute_reply":"2026-09-04T14:46:50.044273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Adding augmentations using *albumentation*.\nFor more info check:\n- [Overview of augmentations](https://albumentations.ai/docs/getting_started/transforms_and_targets/)\n- [Bounding box tutorial](https://albumentations.ai/docs/getting_started/bounding_boxes_augmentation/)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T21:29:09.047218Z","iopub.execute_input":"2024-04-23T21:29:09.04779Z","iopub.status.idle":"2024-04-23T21:29:09.056893Z","shell.execute_reply.started":"2024-04-23T21:29:09.047754Z","shell.execute_reply":"2024-04-23T21:29:09.054798Z"}}},{"cell_type":"code","source":"train_transform = A.Compose(\n    [\n        A.Resize(512, 512),\n\n        A.HorizontalFlip(p=0.5),\n\n        A.ShiftScaleRotate(\n            shift_limit=0.03,\n            scale_limit=0.05,\n            rotate_limit=5,\n            border_mode=0,\n            p=0.5,\n        ),\n\n        A.RandomBrightnessContrast(\n            brightness_limit=0.1,\n            contrast_limit=0.1,\n            p=0.2,\n        ),\n\n        A.Normalize(\n            mean=[0.485, 0.485, 0.485],\n            std=[0.229, 0.229, 0.229],\n        ),\n\n        ToTensorV2(),\n    ],\n    bbox_params=A.BboxParams(\n        format=\"albumentations\",\n        label_fields=[\"class_labels\"],\n        min_visibility=0.2,\n    ),\n)\n\nval_transform = A.Compose(\n    [\n        A.Resize(512, 512),\n\n        A.Normalize(\n            mean=[0.485, 0.485, 0.485],\n            std=[0.229, 0.229, 0.229],\n        ),\n\n        ToTensorV2(),\n    ],\n    bbox_params=A.BboxParams(\n        format=\"albumentations\",\n        label_fields=[\"class_labels\"],\n    ),\n)\n\ndataset = AMIADataset(transform=train_transform, iou_threshold=0.7)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:50.046718Z","iopub.execute_input":"2026-09-04T14:46:50.047053Z","iopub.status.idle":"2026-09-04T14:46:50.217944Z","shell.execute_reply.started":"2026-09-04T14:46:50.047024Z","shell.execute_reply":"2026-09-04T14:46:50.216758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axs = plt.subplots(nrows=1, ncols=3, sharex=True, sharey=True)\n\nfig.set_size_inches(12, 4)\n\nfor i in range(3):\n    image, class_ids, bboxes = dataset[1]\n    plot_bounding_boxes(image, bboxes * 1024, class_ids, ax=axs[i])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:50.219113Z","iopub.execute_input":"2026-09-04T14:46:50.219475Z","iopub.status.idle":"2026-09-04T14:46:50.926610Z","shell.execute_reply.started":"2026-09-04T14:46:50.219452Z","shell.execute_reply":"2026-09-04T14:46:50.925544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training the Model","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# IMAGE-LEVEL MULTI-LABEL TARGETS\n# ============================================================\n\nlabels = (\n    pd.crosstab(\n        train_df[\"image_id\"],\n        train_df[\"class_id\"]\n    )\n    .reindex(columns=range(NUM_CLASSES), fill_value=0)\n    .gt(0)\n    .astype(np.int32)\n)\n\nprint(f\"Shape: {labels.shape}\")\nlabels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:50.927843Z","iopub.execute_input":"2026-09-04T14:46:50.928161Z","iopub.status.idle":"2026-09-04T14:46:51.191816Z","shell.execute_reply.started":"2026-09-04T14:46:50.928135Z","shell.execute_reply":"2026-09-04T14:46:51.190875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def custom_collate_fn(batch):\n    \"\"\"\n    batch: danh sách các tuple (image, target, bboxes) được trả về từ Dataset\n    \"\"\"\n    images = torch.stack([item[0] for item in batch], dim=0)\n    targets = torch.stack([item[1] for item in batch], dim=0)\n    \n    # Giữ bboxes dưới dạng danh sách (list of Tensors) do số lượng box mỗi ảnh khác nhau\n    bboxes = [item[2] for item in batch]\n    \n    return images, targets, bboxes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:51.196783Z","iopub.execute_input":"2026-09-04T14:46:51.197710Z","iopub.status.idle":"2026-09-04T14:46:51.203176Z","shell.execute_reply.started":"2026-09-04T14:46:51.197676Z","shell.execute_reply":"2026-09-04T14:46:51.202464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TRAIN DATALOADER \n\nBATCH_SIZE = 16\n\ntrain_dataset = AMIADataset(\n    partition=\"train\",\n    image_ids=labels.index,\n    labels_df=labels,\n    transform=train_transform,\n)\n\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    num_workers=2,\n    pin_memory=True,\n    persistent_workers=True,\n    collate_fn=custom_collate_fn\n)\n\nimages_batch, targets_batch, bboxes_batch = next(iter(train_loader))\n\nprint(\"Images shape:\", images_batch.shape)\nprint(\"Targets shape:\", targets_batch.shape)\nprint(\"Bboxes type:\", type(bboxes_batch))\n# print(\"First image bboxes shape:\", bboxes_batch[0].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:51.204367Z","iopub.execute_input":"2026-09-04T14:46:51.204914Z","iopub.status.idle":"2026-09-04T14:46:53.031786Z","shell.execute_reply.started":"2026-09-04T14:46:51.204881Z","shell.execute_reply":"2026-09-04T14:46:53.030311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SETUP FOR MULTI-LABEL STRATIFIED K-FOLD\n\nN_SPLITS = 5\nSEED = 42\n\nX = labels.index.to_numpy()\nY = labels.values\n\nmskf = MultilabelStratifiedKFold(n_splits=N_SPLITS, shuffle=True,random_state=SEED)\n\nfolds = []\n\nfor fold, (train_idx, val_idx) in enumerate(mskf.split(X, Y)):\n\n    train_ids = X[train_idx]\n    val_ids = X[val_idx]\n\n    folds.append(\n        {\n            \"fold\": fold,\n            \"train_ids\": train_ids,\n            \"val_ids\": val_ids,\n        }\n    )\n\n    print(\n        f\"Fold {fold}: \"\n        f\"train={len(train_ids)}, \"\n        f\"val={len(val_ids)}\"\n    )\n\nfor fold_info in folds:\n\n    fold = fold_info[\"fold\"]\n    train_ids = fold_info[\"train_ids\"]\n    val_ids = fold_info[\"val_ids\"]\n\n    train_dist = labels.loc[train_ids].mean()\n    val_dist = labels.loc[val_ids].mean()\n\n    comparison = pd.DataFrame({\n        \"train\": train_dist,\n        \"validation\": val_dist,\n    })\n\n    print(f\"\\n===== FOLD {fold} =====\")\n    display(comparison)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:53.034105Z","iopub.execute_input":"2026-09-04T14:46:53.034477Z","iopub.status.idle":"2026-09-04T14:46:53.408764Z","shell.execute_reply.started":"2026-09-04T14:46:53.034437Z","shell.execute_reply":"2026-09-04T14:46:53.407168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ASYMMETRIC LOSS FUNCTION\nclass AsymmetricLoss(nn.Module):\n    def __init__(self, gamma_neg=4, gamma_pos=1, clip=0.05, eps=1e-8):\n        super(AsymmetricLoss, self).__init__()\n        self.gamma_neg = gamma_neg\n        self.gamma_pos = gamma_pos\n        self.clip = clip\n        self.eps = eps\n\n    def forward(self, x, y):\n        \"\"\"\"\n        x: Đầu ra từ mô hình (logits), chưa qua kích hoạt (chưa Sigmoid)\n        y: Nhãn thực tế (ground truth binary vectors), kích thước [batch_size, num_classes]\n        \"\"\"\n        # Tính xác suất qua hàm Sigmoid\n        xs_pos = torch.sigmoid(x)\n        xs_neg = 1 - xs_pos\n\n        # Kỹ thuật Asymmetric Probability Shifting (Clipping) áp dụng cho mẫu tiêu cực\n        if self.clip and self.clip > 0:\n            xs_neg = (xs_neg + self.clip).clamp(max=1)\n\n        # Tính toán tổn thất cơ bản log\n        loss_pos = y * torch.log(xs_pos.clamp(min=self.eps))\n        loss_neg = (1 - y) * torch.log(xs_neg.clamp(min=self.eps))\n\n        # Áp dụng cơ chế tập trung bất đối xứng (Asymmetric Focusing)\n        if self.gamma_pos > 0:\n            loss_pos *= (1 - xs_pos) ** self.gamma_pos\n        if self.gamma_neg > 0:\n            loss_neg *= (1 - xs_neg) ** self.gamma_neg\n\n        # Tổng hợp tổn thất\n        loss = loss_pos + loss_neg\n        return -loss.sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:53.414896Z","iopub.execute_input":"2026-09-04T14:46:53.415207Z","iopub.status.idle":"2026-09-04T14:46:53.425903Z","shell.execute_reply.started":"2026-09-04T14:46:53.415181Z","shell.execute_reply":"2026-09-04T14:46:53.424320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Upgraded: DenseNet121 + ASL (AysmmetricLoss)\nmodel = models.densenet121(weights=models.DenseNet121_Weights.DEFAULT)\n\nNUM_FTRS = model.classifier.in_features \nmodel.classifier = nn.Linear(NUM_FTRS, NUM_CLASSES)\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel.to(device)\n\ncriterion = AsymmetricLoss(gamma_neg=4, gamma_pos=1, clip=0.05)\noptimizer = optim.AdamW(model.parameters(), lr=1e-4, weight_decay=1e-4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:53.427322Z","iopub.execute_input":"2026-09-04T14:46:53.427666Z","iopub.status.idle":"2026-09-04T14:46:54.178752Z","shell.execute_reply.started":"2026-09-04T14:46:53.427615Z","shell.execute_reply":"2026-09-04T14:46:54.177853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CALCULATE METRICS FUNCTION \n\ndef calculate_metrics(y_true, y_prob, threshold=0.5):\n\n    y_pred = (y_prob >= threshold).astype(int)\n\n    metrics = {}\n\n    # -----------------------------------------\n    # AUROC\n    # -----------------------------------------\n\n    try:\n        metrics[\"macro_auc\"] = roc_auc_score(\n            y_true,\n            y_prob,\n            average=\"macro\",\n        )\n    except ValueError:\n        metrics[\"macro_auc\"] = np.nan\n\n    try:\n        metrics[\"micro_auc\"] = roc_auc_score(\n            y_true,\n            y_prob,\n            average=\"micro\",\n        )\n    except ValueError:\n        metrics[\"micro_auc\"] = np.nan\n\n    # -----------------------------------------\n    # mAP\n    # -----------------------------------------\n\n    try:\n        metrics[\"macro_map\"] = average_precision_score(\n            y_true,\n            y_prob,\n            average=\"macro\",\n        )\n    except ValueError:\n        metrics[\"macro_map\"] = np.nan\n\n    # -----------------------------------------\n    # F1\n    # -----------------------------------------\n\n    metrics[\"macro_f1\"] = f1_score(\n        y_true,\n        y_pred,\n        average=\"macro\",\n        zero_division=0,\n    )\n\n    metrics[\"micro_f1\"] = f1_score(\n        y_true,\n        y_pred,\n        average=\"micro\",\n        zero_division=0,\n    )\n\n    metrics[\"macro_precision\"] = precision_score(\n        y_true,\n        y_pred,\n        average=\"macro\",\n        zero_division=0,\n    )\n\n    metrics[\"macro_recall\"] = recall_score(\n        y_true,\n        y_pred,\n        average=\"macro\",\n        zero_division=0,\n    )\n\n    return metrics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:54.180106Z","iopub.execute_input":"2026-09-04T14:46:54.180406Z","iopub.status.idle":"2026-09-04T14:46:54.191502Z","shell.execute_reply.started":"2026-09-04T14:46:54.180381Z","shell.execute_reply":"2026-09-04T14:46:54.190425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PER-CLASS THRESHOLDING FUNCTION\ndef find_best_thresholds(y_true: np.ndarray, y_pred_probs: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Tìm ngưỡng (threshold) tối ưu cho từng class dựa trên F1-Score của Validation set.\n    y_true: (N, num_classes)\n    y_pred_probs: (N, num_classes) - Sigmoid probabilities\n    \"\"\"\n    num_classes = y_true.shape[1]\n    best_thresholds = np.zeros(num_classes)\n    \n    # Duyệt qua từng class\n    for c in range(num_classes):\n        best_f1 = -1.0\n        best_thresh = 0.5  # Mặc định\n        \n        # Grid search ngưỡng từ 0.05 đến 0.95\n        for thresh in np.linspace(0.05, 0.95, 91):\n            preds = (y_pred_probs[:, c] >= thresh).astype(int)\n            f1 = f1_score(y_true[:, c], preds, zero_division=0)\n            \n            if f1 > best_f1:\n                best_f1 = f1\n                best_thresh = thresh\n                \n        best_thresholds[c] = best_thresh\n        \n    return best_thresholds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:54.192816Z","iopub.execute_input":"2026-09-04T14:46:54.193385Z","iopub.status.idle":"2026-09-04T14:46:54.209583Z","shell.execute_reply.started":"2026-09-04T14:46:54.193356Z","shell.execute_reply":"2026-09-04T14:46:54.208431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TRAINING FUNCTION \n\ndef train_one_epoch(model, loader, optimizer, criterion, device):\n\n    model.train()\n\n    running_loss = 0.0\n\n    for images, targets, _ in loader:\n        # Transer images and targets tensors to GPU \n        images = images.to(device, non_blocking=True)\n        targets = targets.float().to(device, non_blocking=True)\n        # Pytorch optimizer tends to accumulate gradients -> Reset gradients after each iteration \n        optimizer.zero_grad(set_to_none=True)\n        # Compute raw logits (No sigmoid activation)\n        logits = model(images)\n        # Compute the loss function (Sigmoid activation + Binary Cross-entropy)\n        loss = criterion(logits, targets)\n        # Backpropagation: compute gradients for all parameters \n        loss.backward()\n        # Update the model parameters using gradient descent \n        optimizer.step()\n\n        running_loss += loss.item() * images.size(0)\n    \n    return running_loss / len(loader.dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:54.211279Z","iopub.execute_input":"2026-09-04T14:46:54.211726Z","iopub.status.idle":"2026-09-04T14:46:54.225464Z","shell.execute_reply.started":"2026-09-04T14:46:54.211686Z","shell.execute_reply":"2026-09-04T14:46:54.224447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# VALIDATION FUNCTION\n\ndef validate(model, loader, criterion, device):\n\n    model.eval()\n\n    running_loss = 0.0\n\n    all_targets = []\n    all_probs = []\n\n    with torch.no_grad():\n\n        for images, targets, _ in loader:\n\n            images = images.to(device, non_blocking=True)\n            targets = targets.float().to(device, non_blocking=True)\n\n            logits = model(images)\n\n            loss = criterion(logits, targets)\n\n            running_loss += (loss.item() * images.size(0))\n\n            probs = torch.sigmoid(logits)\n\n            all_targets.append(targets.cpu().numpy())\n            all_probs.append(probs.cpu().numpy())\n\n    y_true = np.concatenate(all_targets, axis=0)\n    y_prob = np.concatenate(all_probs, axis=0)\n\n    epoch_loss = running_loss / len(loader.dataset)\n    thresholds = find_best_thresholds(y_true, y_prob)\n\n    metrics = calculate_metrics(y_true, y_prob, thresholds)\n    metrics[\"loss\"] = epoch_loss\n\n    return metrics, y_true, y_prob, thresholds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:54.226860Z","iopub.execute_input":"2026-09-04T14:46:54.227186Z","iopub.status.idle":"2026-09-04T14:46:54.239003Z","shell.execute_reply.started":"2026-09-04T14:46:54.227156Z","shell.execute_reply":"2026-09-04T14:46:54.237381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fold = 0\n\ntrain_ids = folds[fold][\"train_ids\"]\nval_ids = folds[fold][\"val_ids\"]\n\ntrain_dataset = AMIADataset(\n    partition=\"train\",\n    image_ids=train_ids,\n    labels_df=labels,\n    transform=train_transform,\n)\n\nval_dataset = AMIADataset(\n    partition=\"train\",\n    image_ids=val_ids,\n    labels_df=labels,\n    transform=val_transform,\n)\n\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=16,\n    shuffle=True,\n    num_workers=2,\n    pin_memory=True,\n    persistent_workers=True,\n    collate_fn=custom_collate_fn\n)\n\nval_loader = DataLoader(\n    val_dataset,\n    batch_size=16,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True,\n    persistent_workers=True,\n    collate_fn=custom_collate_fn\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:54.240555Z","iopub.execute_input":"2026-09-04T14:46:54.241008Z","iopub.status.idle":"2026-09-04T14:46:54.623170Z","shell.execute_reply.started":"2026-09-04T14:46:54.240979Z","shell.execute_reply":"2026-09-04T14:46:54.622270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 5\n\nhistory = []\n\nbest_f1 = 0.0\noptimal_thresholds = np.full(NUM_CLASSES, 0.5)\n\nprint(\"----- BEGIN TRAINING -----\")\nfor epoch in range(EPOCHS):\n    # Freeing up CUDA memory space at start of epoch \n    gc.collect()\n    torch.cuda.empty_cache()\n    \n    # Training the model on train dataset\n    train_loss = train_one_epoch(model, train_loader, optimizer, criterion, device)\n    # Evaluating performance on validation dataset\n    val_metrics, _, _, current_thresholds = validate(model, val_loader, criterion, device)\n    \n    row = {\n        \"epoch\": epoch + 1,\n        \"train_loss\": train_loss,\n        **{\n            f\"val_{k}\": v\n            for k, v in val_metrics.items()\n        },\n    }\n\n    history.append(row)\n\n    print(\n        f\"Epoch {epoch+1:02d} | \"\n        f\"train_loss={train_loss:.4f} | \"\n        f\"val_loss={val_metrics['loss']:.4f} | \"\n        f\"AUROC={val_metrics['macro_auc']:.4f} | \"\n        f\"mAP={val_metrics['macro_map']:.4f} | \"\n        f\"F1={val_metrics['macro_f1']:.4f}\"\n    )\n\n    if val_metrics[\"macro_f1\"] > best_f1:\n\n        best_f1 = val_metrics[\"macro_f1\"]\n        optimal_thresholds = current_thresholds\n\n        torch.save(\n            model.state_dict(),\n            \"best_model_fold0.pth\"\n        )\n\nprint(\"----- TRAINING COMPLETED -----\")\nprint(optimal_thresholds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T14:46:54.624371Z","iopub.execute_input":"2026-09-04T14:46:54.624753Z","iopub.status.idle":"2026-09-04T15:18:39.020555Z","shell.execute_reply.started":"2026-09-04T14:46:54.624730Z","shell.execute_reply":"2026-09-04T15:18:39.018887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TESTING FUNCTION \ndef test(model, loader, device):\n    model.eval()\n    test_predictions = []\n\n    with torch.no_grad():\n        for images, _, _ in loader:  # Hoặc for images in loader tùy thuộc vào DataLoader\n            images = images.to(device)\n            logits = model(images)\n            probs = torch.sigmoid(logits)  # Kích thước: (batch_size, 15)\n\n            # Duyệt qua từng ảnh trong batch\n            for prob_sample in probs:\n                pred_strings = []\n                # Duyệt qua 15 xác suất của ảnh này\n                for class_id, prob in enumerate(prob_sample):\n                    if prob.item() >= optimal_thresholds[class_id]:\n                        # Thêm class_id và độ tin cậy/xác suất (hoặc định dạng theo bài toán)\n                        pred_strings.append(f\"{class_id} {prob.item():.4f}\")\n\n                # Ghép các chuỗi dự đoán của ảnh (nếu không có class nào >= 0.5 thì để trống hoặc \"14 1.0\")\n                if pred_strings:\n                    test_predictions.append(\" \".join(pred_strings))\n                else:\n                    test_predictions.append(\"14 1.0\")  # Ví dụ nhãn 14 là 'No finding'\n\n    return test_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T15:18:39.022490Z","iopub.execute_input":"2026-09-04T15:18:39.023041Z","iopub.status.idle":"2026-09-04T15:18:39.031652Z","shell.execute_reply.started":"2026-09-04T15:18:39.023013Z","shell.execute_reply":"2026-09-04T15:18:39.030274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = read_and_proccess_annotations(\"test\")\n\ntest_dataset = AMIADataset(\n    partition=\"test\",\n    image_ids=test_df[\"image_id\"],\n    labels_df=None,\n    transform=val_transform,\n)\n\ntest_loader = DataLoader(\n    test_dataset,\n    batch_size=16,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True,\n    persistent_workers=True,\n    collate_fn=custom_collate_fn\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T15:18:39.033019Z","iopub.execute_input":"2026-09-04T15:18:39.033396Z","iopub.status.idle":"2026-09-04T15:18:39.210133Z","shell.execute_reply.started":"2026-09-04T15:18:39.033373Z","shell.execute_reply":"2026-09-04T15:18:39.209135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()\ntorch.cuda.empty_cache()\n\ntest_predictions = test(model, test_loader, device)\n\nsubmission = pd.DataFrame({\n    'image_id': test_dataset.image_ids,\n    'PredictionString': test_predictions\n})\n\nprint(submission.head())\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-04T15:18:39.211310Z","iopub.execute_input":"2026-09-04T15:18:39.212250Z","iopub.status.idle":"2026-09-04T15:27:06.142157Z","shell.execute_reply.started":"2026-09-04T15:18:39.212223Z","shell.execute_reply":"2026-09-04T15:27:06.140721Z"}},"outputs":[],"execution_count":null}]}