{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"},{"sourceId":7283396,"sourceType":"datasetVersion","datasetId":4223288}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Project description","metadata":{}},{"cell_type":"markdown","source":"## Data Description","metadata":{}},{"cell_type":"markdown","source":"## Installing and importing libraries","metadata":{}},{"cell_type":"code","source":"!pip install torchsummary\n!pip install segmentation_models_pytorch\n# !pip install torch torchvision torchaudio -f https://download.pytorch.org/whl/cu110/torch_stable.html","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:24:34.586977Z","iopub.execute_input":"2024-01-30T09:24:34.588118Z","iopub.status.idle":"2024-01-30T09:25:12.851056Z","shell.execute_reply.started":"2024-01-30T09:24:34.588047Z","shell.execute_reply":"2024-01-30T09:25:12.849937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nfrom tqdm import tqdm\nfrom pathlib import Path\n\nfrom glob import glob\nimport random\n\nimport warnings\nimport time\nfrom collections import Counter\n\nfrom skimage import io\nimport tifffile as tiff\nfrom skimage import measure\nimport seaborn as sns\nfrom matplotlib.lines import Line2D\nimport matplotlib.image as mpimg\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nimport torch\nimport torch.nn.init as init\nimport torch.cuda.amp as amp\nfrom torch.cuda.amp import autocast\nfrom torchvision import transforms\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn as nn\nfrom torch.optim import Adam\nimport segmentation_models_pytorch as smp\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n\nimport imgaug.augmenters as iaa\nfrom torchsummary import summary\n\nfrom torch.utils.data import ConcatDataset","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-30T09:25:12.853911Z","iopub.execute_input":"2024-01-30T09:25:12.854327Z","iopub.status.idle":"2024-01-30T09:25:37.175562Z","shell.execute_reply.started":"2024-01-30T09:25:12.854291Z","shell.execute_reply":"2024-01-30T09:25:37.174377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input')\n\nfrom metric import metric","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:37.176879Z","iopub.execute_input":"2024-01-30T09:25:37.177656Z","iopub.status.idle":"2024-01-30T09:25:38.575517Z","shell.execute_reply.started":"2024-01-30T09:25:37.177620Z","shell.execute_reply":"2024-01-30T09:25:38.574345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"# Disable warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.578027Z","iopub.execute_input":"2024-01-30T09:25:38.578409Z","iopub.status.idle":"2024-01-30T09:25:38.583404Z","shell.execute_reply.started":"2024-01-30T09:25:38.578377Z","shell.execute_reply":"2024-01-30T09:25:38.582257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class f:    \n    BOLD = \"\\033[1m\"     # Bold text\n    ITALIC = \"\\033[3m\"   # Italic text\n    END = \"\\033[0m\"      # Reset style","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.584854Z","iopub.execute_input":"2024-01-30T09:25:38.585285Z","iopub.status.idle":"2024-01-30T09:25:38.600125Z","shell.execute_reply.started":"2024-01-30T09:25:38.585245Z","shell.execute_reply":"2024-01-30T09:25:38.599161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"facecolor='lightgray'\n# gridcolor='gray'\n\ncolors = ['pink', 'steelblue', 'hotpink','lightgreen','gray','salmon','gold', 'seagreen', 'skyblue', 'orchid']","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.601274Z","iopub.execute_input":"2024-01-30T09:25:38.602442Z","iopub.status.idle":"2024-01-30T09:25:38.611412Z","shell.execute_reply.started":"2024-01-30T09:25:38.602396Z","shell.execute_reply":"2024-01-30T09:25:38.610463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"def explore_data(dataframes):\n    for name, df in dataframes.items():\n        print(f\"\\n{name} Data:\")\n        print(f\"Shape:\\n{df.shape}\")\n        print(f\"\\nInfo:\")\n        df.info()\n        print(f\"\\nDuplicates:\")\n        duplicates = df[df.duplicated()]\n        print(duplicates)\n        print(f\"\\nMissing Values:\")\n        missing_values = df.isnull().sum()\n        print(missing_values[missing_values > 0])","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.613782Z","iopub.execute_input":"2024-01-30T09:25:38.614640Z","iopub.status.idle":"2024-01-30T09:25:38.624302Z","shell.execute_reply.started":"2024-01-30T09:25:38.614597Z","shell.execute_reply":"2024-01-30T09:25:38.623460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_and_display_samples(image_paths, title):\n    plt.figure(figsize=(15, 5))\n    for i in range(3):\n        plt.subplot(1, 3, i + 1)\n        img = tiff.imread(image_paths[i])\n        plt.imshow(img, cmap='gray')\n        plt.title(f'{title} Sample {i + 1}')\n        plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.625790Z","iopub.execute_input":"2024-01-30T09:25:38.626667Z","iopub.status.idle":"2024-01-30T09:25:38.643935Z","shell.execute_reply.started":"2024-01-30T09:25:38.626624Z","shell.execute_reply":"2024-01-30T09:25:38.642700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image_sizes(image_paths):\n    sizes = []\n    for img_path in image_paths:\n        img = tiff.imread(img_path)\n        sizes.append(img.shape)\n    return sizes","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.645184Z","iopub.execute_input":"2024-01-30T09:25:38.647379Z","iopub.status.idle":"2024-01-30T09:25:38.655541Z","shell.execute_reply.started":"2024-01-30T09:25:38.647335Z","shell.execute_reply":"2024-01-30T09:25:38.654642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_image_and_mask(image_folder, mask_folder, subdataset):\n    image_files = os.listdir(image_folder)\n    mask_files = os.listdir(mask_folder)\n    \n    image_path = os.path.join(image_folder, image_files[0])\n    mask_path = os.path.join(mask_folder, mask_files[0])\n\n    image = cv2.imread(image_path, cv2.IMREAD_UNCHANGED)\n    mask = cv2.imread(mask_path, cv2.IMREAD_UNCHANGED)\n    \n    plt.figure(figsize=(10, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.imshow(image, cmap='gray')\n    plt.title('Original Image')\n    \n    plt.subplot(1, 2, 2)\n    plt.imshow(mask, cmap='gray')\n    plt.title('Mask')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.660317Z","iopub.execute_input":"2024-01-30T09:25:38.660732Z","iopub.status.idle":"2024-01-30T09:25:38.670157Z","shell.execute_reply.started":"2024-01-30T09:25:38.660691Z","shell.execute_reply":"2024-01-30T09:25:38.668950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_mask_statistics(mask_folder):\n    mask_files = os.listdir(mask_folder)\n    mask_areas = []\n\n    for mask_file in tqdm(mask_files, desc='Calculating mask statistics'):\n        mask_path = os.path.join(mask_folder, mask_file)\n        mask = cv2.imread(mask_path, cv2.IMREAD_GRAYSCALE)\n\n        mask_area = np.sum(mask == 255)\n        mask_areas.append(mask_area)\n\n    return mask_areas","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.671899Z","iopub.execute_input":"2024-01-30T09:25:38.672589Z","iopub.status.idle":"2024-01-30T09:25:38.681776Z","shell.execute_reply.started":"2024-01-30T09:25:38.672557Z","shell.execute_reply":"2024-01-30T09:25:38.680954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_mask_statistics(mask_areas):\n    plt.figure(figsize=(12, 6))\n\n    plt.subplot(1, 2, 1)\n    plt.hist(mask_areas, bins=20, color='skyblue', edgecolor='black')\n    plt.title('Distribution of Mask Areas')\n    plt.xlabel('Mask Area')\n    plt.ylabel('Frequency')\n\n    plt.subplot(1, 2, 2)\n    plt.boxplot(mask_areas, vert=False)\n    plt.title('Boxplot of Mask Areas')\n    plt.xlabel('Mask Area')\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.683248Z","iopub.execute_input":"2024-01-30T09:25:38.683843Z","iopub.status.idle":"2024-01-30T09:25:38.694251Z","shell.execute_reply.started":"2024-01-30T09:25:38.683811Z","shell.execute_reply":"2024-01-30T09:25:38.693188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_class_balance(mask_folder, num_classes):\n    mask_files = os.listdir(mask_folder)\n    class_counts = np.zeros(num_classes)\n\n    for mask_file in tqdm(mask_files, desc='Calculating class balance'):\n        mask_path = os.path.join(mask_folder, mask_file)\n        mask = cv2.imread(mask_path, cv2.IMREAD_GRAYSCALE)\n\n        for class_id in range(num_classes):\n            class_counts[class_id] += np.sum(mask == class_id)\n\n    return class_counts\n","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.695738Z","iopub.execute_input":"2024-01-30T09:25:38.696142Z","iopub.status.idle":"2024-01-30T09:25:38.712397Z","shell.execute_reply.started":"2024-01-30T09:25:38.696098Z","shell.execute_reply":"2024-01-30T09:25:38.711281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_class_balance(class_counts):\n    num_classes = len(class_counts)\n    classes = [f'Class {i}' for i in range(num_classes)]\n\n    plt.bar(classes, class_counts, color=colors)\n    plt.title('Class Balance')\n    plt.xlabel('Class')\n    plt.ylabel('Pixel Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.714159Z","iopub.execute_input":"2024-01-30T09:25:38.714539Z","iopub.status.idle":"2024-01-30T09:25:38.725241Z","shell.execute_reply.started":"2024-01-30T09:25:38.714508Z","shell.execute_reply":"2024-01-30T09:25:38.724149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data","metadata":{}},{"cell_type":"code","source":"base_path = '/kaggle/input/blood-vessel-segmentation/'","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.726673Z","iopub.execute_input":"2024-01-30T09:25:38.727011Z","iopub.status.idle":"2024-01-30T09:25:38.737554Z","shell.execute_reply.started":"2024-01-30T09:25:38.726983Z","shell.execute_reply":"2024-01-30T09:25:38.736457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(os.path.join(base_path, 'sample_submission.csv'))\ntrain_rles = pd.read_csv(os.path.join(base_path, 'train_rles.csv'))","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:38.738856Z","iopub.execute_input":"2024-01-30T09:25:38.739190Z","iopub.status.idle":"2024-01-30T09:25:40.032310Z","shell.execute_reply.started":"2024-01-30T09:25:38.739161Z","shell.execute_reply":"2024-01-30T09:25:40.030945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train Data:\")\ndisplay(train_rles.head())\n\nprint(\"\\nSubmission Data:\")\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:40.035673Z","iopub.execute_input":"2024-01-30T09:25:40.036752Z","iopub.status.idle":"2024-01-30T09:25:40.065899Z","shell.execute_reply.started":"2024-01-30T09:25:40.036704Z","shell.execute_reply":"2024-01-30T09:25:40.064812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_paths = {\n    'train': {\n        'kidney_1_dense': {\n            'images': os.path.join(base_path, 'train/kidney_1_dense/images'),\n            'labels': os.path.join(base_path, 'train/kidney_1_dense/labels')\n        },\n        'kidney_1_voi': {\n            'images': os.path.join(base_path, 'train/kidney_1_voi/images'),\n            'labels': os.path.join(base_path, 'train/kidney_1_voi/labels')\n        },\n        'kidney_2': {\n            'images': os.path.join(base_path, 'train/kidney_2/images'),\n            'labels': os.path.join(base_path, 'train/kidney_2/labels')\n        },\n        'kidney_3_dense': {\n            'labels': os.path.join(base_path, 'train/kidney_3_dense/labels')\n        },\n        'kidney_3_sparse': {\n            'images': os.path.join(base_path, 'train/kidney_3_sparse/images'),\n            'labels': os.path.join(base_path, 'train/kidney_3_sparse/labels')\n        }\n    },\n    'test': {\n        'kidney_5': os.path.join(base_path, 'test/kidney_5/images'),\n        'kidney_6': os.path.join(base_path, 'test/kidney_6/images')\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:40.067216Z","iopub.execute_input":"2024-01-30T09:25:40.067615Z","iopub.status.idle":"2024-01-30T09:25:40.076309Z","shell.execute_reply.started":"2024-01-30T09:25:40.067586Z","shell.execute_reply":"2024-01-30T09:25:40.075157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset, subdatasets in data_paths['train'].items():\n    for subdataset, paths in subdatasets.items():\n        if isinstance(paths, dict):\n            image_files = os.listdir(paths['images'])\n            image_paths = [os.path.join(paths['images'], img) for img in image_files]\n            load_and_display_samples(image_paths, f'Train - {dataset} - {subdataset}')\n        else:\n            image_files = os.listdir(paths)\n            image_paths = [os.path.join(paths, img) for img in image_files]\n            load_and_display_samples(image_paths, f'Train - {dataset}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:40.078199Z","iopub.execute_input":"2024-01-30T09:25:40.078976Z","iopub.status.idle":"2024-01-30T09:25:50.685377Z","shell.execute_reply.started":"2024-01-30T09:25:40.078935Z","shell.execute_reply":"2024-01-30T09:25:50.684355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset, paths in data_paths['test'].items():\n    if isinstance(paths, dict):\n        image_files = os.listdir(paths['images'])\n        image_paths = [os.path.join(paths['images'], img) for img in image_files]\n        load_and_display_samples(image_paths, f'Test - {dataset}')\n    else:\n        image_files = os.listdir(paths)\n        image_paths = [os.path.join(paths, img) for img in image_files]\n        load_and_display_samples(image_paths, f'Test - {dataset}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:50.686961Z","iopub.execute_input":"2024-01-30T09:25:50.687314Z","iopub.status.idle":"2024-01-30T09:25:52.603129Z","shell.execute_reply.started":"2024-01-30T09:25:50.687285Z","shell.execute_reply":"2024-01-30T09:25:52.602264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's check the received data","metadata":{}},{"cell_type":"code","source":"explore_data(train_rles)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:52.604413Z","iopub.execute_input":"2024-01-30T09:25:52.605343Z","iopub.status.idle":"2024-01-30T09:25:52.676511Z","shell.execute_reply.started":"2024-01-30T09:25:52.605310Z","shell.execute_reply":"2024-01-30T09:25:52.675262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we have a DataFrame with two columns representing identifiers (\"id\") and data in RLE format (\"rle\"). There are duplicates found in column \"rle\", but these are only binary values  \n\n`Shape:` (7429, 2)  \n\n`Information:`  \n\n`Index:` RangeIndex from 0 to 7428  \n`Number of columns:` 2 (id, rle)  \n`Number of non-blank values:` 7429 for each column  \n`Column data types:` object  \n\n`Duplicates:`  \nThere are duplicates in column \"rle\" where the values are \"1 0\"  \nThe total number of unique values for \"rle\" is not specified  \n\n`Missing Values:`\nMissing in both columns  ","metadata":{}},{"cell_type":"markdown","source":"## Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"### Let's look at the size of the images in the folders of the Train section","metadata":{}},{"cell_type":"code","source":"image_paths_train = []\nfor dataset, subdatasets in data_paths['train'].items():\n    for subdataset, paths in subdatasets.items():\n        if isinstance(paths, dict) and 'images' in paths:\n            image_files = os.listdir(paths['images'])\n            image_paths_train.extend([os.path.join(paths['images'], img) for img in image_files])","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:52.678075Z","iopub.execute_input":"2024-01-30T09:25:52.678579Z","iopub.status.idle":"2024-01-30T09:25:52.686588Z","shell.execute_reply.started":"2024-01-30T09:25:52.678525Z","shell.execute_reply":"2024-01-30T09:25:52.685491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sizes = get_image_sizes(image_paths_train)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:52.687943Z","iopub.execute_input":"2024-01-30T09:25:52.688315Z","iopub.status.idle":"2024-01-30T09:25:52.698487Z","shell.execute_reply.started":"2024-01-30T09:25:52.688284Z","shell.execute_reply":"2024-01-30T09:25:52.697438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 6))\n\nbar_width = 0.2\nbar_positions = [i * bar_width for i in range(len(data_paths['train']))]\n\nlegend_patches = []\n\ndef generate_color():\n    return np.random.rand(3,)\n\nfor i, (subdataset, paths) in enumerate(data_paths['train'].items()):\n    if 'images' in paths:\n        colors = generate_color()\n        image_paths = [os.path.join(paths['images'], img) for img in os.listdir(paths['images'])]\n        subdataset_sizes = [tiff.imread(img_path).shape for img_path in image_paths]\n        subdataset_sizes = [size[0] + size[1] for size in subdataset_sizes]\n        subdataset_positions = [pos + bar_positions[i] for pos in range(len(subdataset_sizes))]\n        ax.bar(subdataset_positions, subdataset_sizes, width=bar_width, label=f'{subdataset}', color=colors)\n\n        legend_patches.append(Line2D([0], [0], marker='s', color='w', label=f'{subdataset}', markerfacecolor=colors, markersize=10))\n\nax.set_xticks([pos + (len(data_paths['train']) - 1) * bar_width / 2 for pos in range(len(train_sizes))])\nax.set_xticklabels([f'Subdataset {i}' for i in range(1, len(train_sizes) + 1)])\n\nax.set_ylabel('Image Size (Width + Height)')\nax.set_title('Image Sizes in Different Subdatasets (Train)')\n\nax.legend(handles=legend_patches)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:25:52.700352Z","iopub.execute_input":"2024-01-30T09:25:52.700835Z","iopub.status.idle":"2024-01-30T09:31:59.445835Z","shell.execute_reply.started":"2024-01-30T09:25:52.700792Z","shell.execute_reply":"2024-01-30T09:31:59.444348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `kidney_1_voi` set has the largest image size in each of the two folders with **1397 images** and they have the largest size  \n\nNext, `kidney_3_sparse`, it also contains two folders of **1035 images** and their size is the second after `kidney_1_voi`    \n\nNext comes `kidney_2` despite the fact that both folders contain **2217 images**    \n\nThe `kidney_1_dense` set is the smallest in terms of image size, although both of its folders also have a larger number of images, **2279 in each**  \n\n`kidney_3_dense` contains only masks and is not included in this comparison  ","metadata":{}},{"cell_type":"markdown","source":"### Let's look at examples of images and their corresponding masks. We will also make sure that the masks highlight the vessels correctly.","metadata":{}},{"cell_type":"code","source":"subdataset = 'kidney_1_dense'\nvisualize_image_and_mask(data_paths['train'][subdataset]['images'], \n                         data_paths['train'][subdataset]['labels'], subdataset)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:31:59.448917Z","iopub.execute_input":"2024-01-30T09:31:59.449823Z","iopub.status.idle":"2024-01-30T09:32:00.232681Z","shell.execute_reply.started":"2024-01-30T09:31:59.449772Z","shell.execute_reply":"2024-01-30T09:32:00.231286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subdataset = 'kidney_1_voi'\nvisualize_image_and_mask(data_paths['train'][subdataset]['images'], \n                         data_paths['train'][subdataset]['labels'], subdataset)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:32:00.234690Z","iopub.execute_input":"2024-01-30T09:32:00.235855Z","iopub.status.idle":"2024-01-30T09:32:01.556968Z","shell.execute_reply.started":"2024-01-30T09:32:00.235809Z","shell.execute_reply":"2024-01-30T09:32:01.555747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subdataset = 'kidney_2'\nvisualize_image_and_mask(data_paths['train'][subdataset]['images'], \n                         data_paths['train'][subdataset]['labels'], subdataset)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:32:01.558522Z","iopub.execute_input":"2024-01-30T09:32:01.558939Z","iopub.status.idle":"2024-01-30T09:32:02.294888Z","shell.execute_reply.started":"2024-01-30T09:32:01.558904Z","shell.execute_reply":"2024-01-30T09:32:02.293588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subdataset = 'kidney_3_sparse'\nvisualize_image_and_mask(data_paths['train'][subdataset]['images'], \n                         data_paths['train'][subdataset]['labels'], subdataset)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:32:02.303952Z","iopub.execute_input":"2024-01-30T09:32:02.304865Z","iopub.status.idle":"2024-01-30T09:32:03.228850Z","shell.execute_reply.started":"2024-01-30T09:32:02.304821Z","shell.execute_reply":"2024-01-30T09:32:03.227451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We see that the masks match the images quite accurately in all sets","metadata":{}},{"cell_type":"markdown","source":"### Let's look at the statistics of masks: what part of the image is usually covered with vessels, what are their sizes, how are they distributed across the images. We will also calculate statistical metrics such as mean, median, standard deviation","metadata":{}},{"cell_type":"code","source":"subdataset = 'kidney_1_dense'\nmask_folder = data_paths['train'][subdataset]['labels']\nmask_areas = calculate_mask_statistics(mask_folder)\nvisualize_mask_statistics(mask_areas)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:32:03.230306Z","iopub.execute_input":"2024-01-30T09:32:03.230685Z","iopub.status.idle":"2024-01-30T09:32:54.752976Z","shell.execute_reply.started":"2024-01-30T09:32:03.230655Z","shell.execute_reply":"2024-01-30T09:32:54.751718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subdataset = 'kidney_1_voi'\nmask_folder = data_paths['train'][subdataset]['labels']\nmask_areas = calculate_mask_statistics(mask_folder)\nvisualize_mask_statistics(mask_areas)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:32:54.754519Z","iopub.execute_input":"2024-01-30T09:32:54.754910Z","iopub.status.idle":"2024-01-30T09:34:12.225378Z","shell.execute_reply.started":"2024-01-30T09:32:54.754877Z","shell.execute_reply":"2024-01-30T09:34:12.223934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subdataset = 'kidney_2'\nmask_folder = data_paths['train'][subdataset]['labels']\nmask_areas = calculate_mask_statistics(mask_folder)\nvisualize_mask_statistics(mask_areas)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:34:12.227607Z","iopub.execute_input":"2024-01-30T09:34:12.228125Z","iopub.status.idle":"2024-01-30T09:35:09.723203Z","shell.execute_reply.started":"2024-01-30T09:34:12.228057Z","shell.execute_reply":"2024-01-30T09:35:09.721821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subdataset = 'kidney_3_sparse'\nmask_folder = data_paths['train'][subdataset]['labels']\nmask_areas = calculate_mask_statistics(mask_folder)\nvisualize_mask_statistics(mask_areas)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:35:09.725477Z","iopub.execute_input":"2024-01-30T09:35:09.726009Z","iopub.status.idle":"2024-01-30T09:35:46.832250Z","shell.execute_reply.started":"2024-01-30T09:35:09.725929Z","shell.execute_reply":"2024-01-30T09:35:46.830895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's look at Class Analysis: If there are several classes of vessels in a set, let's evaluate the balance between the classes. We need to make sure that there is enough data for each class.","metadata":{}},{"cell_type":"markdown","source":"#### `kidney_1_dense`","metadata":{}},{"cell_type":"code","source":"mask_folder = '/kaggle/input/blood-vessel-segmentation/train/kidney_1_dense/labels'\nmask_files = os.listdir(mask_folder)\nmask_sizes = []\n\nfor mask_file in mask_files:\n    mask_path = os.path.join(mask_folder, mask_file)\n    mask = cv2.imread(mask_path, cv2.IMREAD_UNCHANGED)\n\n    if mask is not None:\n        mask_size = np.sum(mask == 255)\n        mask_sizes.append(mask_size)\n    else:\n         print(f'Failed to read mask on path: {mask_path}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:35:46.833900Z","iopub.execute_input":"2024-01-30T09:35:46.834313Z","iopub.status.idle":"2024-01-30T09:35:58.758073Z","shell.execute_reply.started":"2024-01-30T09:35:46.834277Z","shell.execute_reply":"2024-01-30T09:35:58.757139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"size_counter = Counter(mask_sizes)\n\nbar_width = 50  \nbar_alpha = 0.5\n\nfor size, count in size_counter.items():\n    random_color = np.random.rand(3,)\n    plt.bar(size, count, color=[random_color], width=bar_width, alpha=bar_alpha)\n\nplt.xlabel('Mask size')\nplt.ylabel('Number of masks')\nplt.title('Distribution of mask sizes')\nplt.ylim(0, 5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:35:58.759323Z","iopub.execute_input":"2024-01-30T09:35:58.759828Z","iopub.status.idle":"2024-01-30T09:36:02.967371Z","shell.execute_reply.started":"2024-01-30T09:35:58.759797Z","shell.execute_reply":"2024-01-30T09:36:02.966100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### `kidney_1_voi`","metadata":{}},{"cell_type":"code","source":"mask_folder = '/kaggle/input/blood-vessel-segmentation/train/kidney_1_voi/labels'\nmask_files = os.listdir(mask_folder)\nmask_sizes = []\n\nfor mask_file in mask_files:\n    mask_path = os.path.join(mask_folder, mask_file)\n    mask = cv2.imread(mask_path, cv2.IMREAD_UNCHANGED)\n\n    if mask is not None:\n        mask_size = np.sum(mask == 255)\n        mask_sizes.append(mask_size)\n    else:\n         print(f'Failed to read mask on path: {mask_path}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:36:02.969027Z","iopub.execute_input":"2024-01-30T09:36:02.970290Z","iopub.status.idle":"2024-01-30T09:36:22.788895Z","shell.execute_reply.started":"2024-01-30T09:36:02.970247Z","shell.execute_reply":"2024-01-30T09:36:22.787599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"size_counter = Counter(mask_sizes)\n\nbar_width = 50  \nbar_alpha = 0.5\n\nfor size, count in size_counter.items():\n    random_color = np.random.rand(3,)\n    plt.bar(size, count, color=[random_color], width=bar_width, alpha=bar_alpha)\n\nplt.xlabel('Mask size')\nplt.ylabel('Number of masks')\nplt.title('Distribution of mask sizes')\nplt.ylim(0, 2.5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:36:22.790782Z","iopub.execute_input":"2024-01-30T09:36:22.791278Z","iopub.status.idle":"2024-01-30T09:36:26.460601Z","shell.execute_reply.started":"2024-01-30T09:36:22.791232Z","shell.execute_reply":"2024-01-30T09:36:26.459539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### `kidney_2`","metadata":{}},{"cell_type":"code","source":"mask_folder = '/kaggle/input/blood-vessel-segmentation/train/kidney_2/labels'\nmask_files = os.listdir(mask_folder)\nmask_sizes = []\n\nfor mask_file in mask_files:\n    mask_path = os.path.join(mask_folder, mask_file)\n    mask = cv2.imread(mask_path, cv2.IMREAD_UNCHANGED)\n\n    if mask is not None:\n        mask_size = np.sum(mask == 255)\n        mask_sizes.append(mask_size)\n    else:\n         print(f'Failed to read mask on path: {mask_path}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:36:26.462010Z","iopub.execute_input":"2024-01-30T09:36:26.462422Z","iopub.status.idle":"2024-01-30T09:36:40.964744Z","shell.execute_reply.started":"2024-01-30T09:36:26.462390Z","shell.execute_reply":"2024-01-30T09:36:40.963525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"size_counter = Counter(mask_sizes)\n\nbar_width = 50  \nbar_alpha = 0.5\n\nfor size, count in size_counter.items():\n    random_color = np.random.rand(3,)\n    plt.bar(size, count, color=[random_color], width=bar_width, alpha=bar_alpha)\n\nplt.xlabel('Mask size')\nplt.ylabel('Number of masks')\nplt.title('Distribution of mask sizes')\nplt.ylim(0, 5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:36:40.966454Z","iopub.execute_input":"2024-01-30T09:36:40.966950Z","iopub.status.idle":"2024-01-30T09:36:44.846286Z","shell.execute_reply.started":"2024-01-30T09:36:40.966904Z","shell.execute_reply":"2024-01-30T09:36:44.844863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### `kidney_3_dense`","metadata":{}},{"cell_type":"code","source":"mask_folder = '/kaggle/input/blood-vessel-segmentation/train/kidney_3_dense/labels'\nmask_files = os.listdir(mask_folder)\nmask_sizes = []\n\nfor mask_file in mask_files:\n    mask_path = os.path.join(mask_folder, mask_file)\n    mask = cv2.imread(mask_path, cv2.IMREAD_UNCHANGED)\n\n    if mask is not None:\n        mask_size = np.sum(mask == 255)\n        mask_sizes.append(mask_size)\n    else:\n         print(f'Failed to read mask on path: {mask_path}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:36:44.847995Z","iopub.execute_input":"2024-01-30T09:36:44.848504Z","iopub.status.idle":"2024-01-30T09:37:01.764659Z","shell.execute_reply.started":"2024-01-30T09:36:44.848457Z","shell.execute_reply":"2024-01-30T09:37:01.763427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"size_counter = Counter(mask_sizes)\n\nbar_width = 50  \nbar_alpha = 0.5\n\nfor size, count in size_counter.items():\n    random_color = np.random.rand(3,)\n    plt.bar(size, count, color=[random_color], width=bar_width, alpha=bar_alpha)\n\nplt.xlabel('Mask size')\nplt.ylabel('Number of masks')\nplt.title('Distribution of mask sizes')\nplt.ylim(0, 5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:01.766230Z","iopub.execute_input":"2024-01-30T09:37:01.766626Z","iopub.status.idle":"2024-01-30T09:37:02.954655Z","shell.execute_reply.started":"2024-01-30T09:37:01.766593Z","shell.execute_reply":"2024-01-30T09:37:02.953483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### `kidney_3_sparse`","metadata":{}},{"cell_type":"code","source":"mask_folder = '/kaggle/input/blood-vessel-segmentation/train/kidney_3_sparse/labels'\nmask_files = os.listdir(mask_folder)\nmask_sizes = []\n\nfor mask_file in mask_files:\n    mask_path = os.path.join(mask_folder, mask_file)\n    mask = cv2.imread(mask_path, cv2.IMREAD_UNCHANGED)\n\n    if mask is not None:\n        mask_size = np.sum(mask == 255)\n        mask_sizes.append(mask_size)\n    else:\n         print(f'Failed to read mask on path: {mask_path}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:02.956428Z","iopub.execute_input":"2024-01-30T09:37:02.956903Z","iopub.status.idle":"2024-01-30T09:37:13.857540Z","shell.execute_reply.started":"2024-01-30T09:37:02.956859Z","shell.execute_reply":"2024-01-30T09:37:13.855990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"size_counter = Counter(mask_sizes)\n\nbar_width = 50  \nbar_alpha = 0.5\n\nfor size, count in size_counter.items():\n    random_color = np.random.rand(3,)\n    plt.bar(size, count, color=[random_color], width=bar_width, alpha=bar_alpha)\n\nplt.xlabel('Mask size')\nplt.ylabel('Number of masks')\nplt.title('Distribution of mask sizes')\nplt.ylim(0, 5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:13.859422Z","iopub.execute_input":"2024-01-30T09:37:13.859824Z","iopub.status.idle":"2024-01-30T09:37:15.918640Z","shell.execute_reply.started":"2024-01-30T09:37:13.859790Z","shell.execute_reply":"2024-01-30T09:37:15.917720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The uneven distribution of mask sizes, as well as the presence of a small number of instances for some sizes, can introduce certain difficulties in training a segmentation model. Here are some thoughts:  \n\n>Class imbalance: If the sizes of objects in the images vary significantly, this can lead to imbalanced classes. Small objects may receive less attention during training, which may result in the model being less accurate at identifying small objects  \n\n>Selecting a Loss Function: Unbalanced object sizes may require the use of a loss function that takes into account class or pixel weights. For example, a loss function that takes into account the frequency of each size can help the model learn better on smaller objects  \n\n>Heterogeneity problem: If some object sizes are represented by only one or a few instances, this can lead to a heterogeneity problem in the data. The model may underestimate or overestimate such dimensions due to the limited number of training examples  \n\n>Image Scaling: Depending on the model architecture, images or objects may need to be scaled to accommodate different sizes. This can help the model learn better on objects of different scales  \n\n>Data Augmentation: Using various data augmentation techniques, such as resizing, rotating, and cropping, can help enhance model training on objects of varying sizes  ","metadata":{}},{"cell_type":"markdown","source":"## Model training","metadata":{}},{"cell_type":"code","source":"# # The configuration class\n\n# class CFG:\n#     model_name = 'Unet'\n#     backbone = 'se_resnext50_32x4d'\n#     in_chans = 1\n#     image_size = 1024\n#     input_size = 1024\n#     target_size = 1\n#     chopping_percentile = 1e-3\n#     batch = 16\n#     drop_edge_pixel = 2\n#     axis_second_model = 0\n#     axis_w = [1, 0.5, 0.5]  \n#     tile_size = 256\n#     stride = 256\n#     th_percentile = 95","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:15.920001Z","iopub.execute_input":"2024-01-30T09:37:15.920701Z","iopub.status.idle":"2024-01-30T09:37:15.927730Z","shell.execute_reply.started":"2024-01-30T09:37:15.920664Z","shell.execute_reply":"2024-01-30T09:37:15.926186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to normalize input tensor\n\n# def norm_with_clip(x, smooth=1e-5):\n#     dim = list(range(1, x.ndim))\n#     mean = x.mean(dim=dim, keepdim=True)\n#     std = x.std(dim=dim, keepdim=True)\n#     x = (x - mean) / (std + smooth)\n#     x[x > 5] = (x[x > 5] - 5) * 1e-3 + 5\n#     x[x < -3] = (x[x < -3] + 3) * 1e-3 - 3\n#     return x","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:15.929476Z","iopub.execute_input":"2024-01-30T09:37:15.930019Z","iopub.status.idle":"2024-01-30T09:37:15.943526Z","shell.execute_reply.started":"2024-01-30T09:37:15.929964Z","shell.execute_reply":"2024-01-30T09:37:15.942211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to resize image to 1024x1024 with rotation\n\n# def to_1024(img, image_size=1024):\n#     if image_size > img.shape[1]:\n#         img = np.rot90(img)\n#         start1 = (CFG.image_size - img.shape[0]) // 2\n#         top = img[0: start1, 0: img.shape[1]]\n#         bottom = img[img.shape[0] - start1: img.shape[0], 0: img.shape[1]]\n#         img_result = np.concatenate((top, img, bottom), axis=0)\n#         img_result = np.rot90(img_result)\n#         img_result = np.rot90(img_result)\n#         img_result = np.rot90(img_result)\n#     else:\n#         img_result = img\n#     return img_result","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:15.945391Z","iopub.execute_input":"2024-01-30T09:37:15.945945Z","iopub.status.idle":"2024-01-30T09:37:15.959746Z","shell.execute_reply.started":"2024-01-30T09:37:15.945894Z","shell.execute_reply":"2024-01-30T09:37:15.958309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to load and normalize data\n\n# def load_and_normalize_data(paths):\n#     data = []\n#     for path in paths:\n#         img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n#         img = to_1024(img)\n#         img = torch.from_numpy(img)\n#         img = img.to(torch.float32)\n#         img = norm_with_clip(img).reshape(1, *img.shape)\n#         data.append(img)\n#     return torch.cat(data, dim=0)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:15.962122Z","iopub.execute_input":"2024-01-30T09:37:15.962751Z","iopub.status.idle":"2024-01-30T09:37:15.980732Z","shell.execute_reply.started":"2024-01-30T09:37:15.962705Z","shell.execute_reply":"2024-01-30T09:37:15.979050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Dataset class for loading images\n\n# class Data_loader(Dataset):\n#     def __init__(self, paths):\n#         self.paths = paths\n\n#     def __len__(self):\n#         return len(self.paths)\n\n#     def __getitem__(self, index):\n#         img = cv2.imread(self.paths[index], cv2.IMREAD_GRAYSCALE)\n#         img = to_1024(img)\n#         img = torch.from_numpy(img)\n#         img = img.to(torch.float32)\n#         img = norm_with_clip(img).reshape(1, *img.shape)\n#         return img","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:15.982951Z","iopub.execute_input":"2024-01-30T09:37:15.984695Z","iopub.status.idle":"2024-01-30T09:37:15.998551Z","shell.execute_reply.started":"2024-01-30T09:37:15.984569Z","shell.execute_reply":"2024-01-30T09:37:15.997215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to prepare data loaders\n\n# def prepare_data_loaders(train_paths, batch_size):\n#     train_loader = DataLoader(Data_loader(train_paths), batch_size=batch_size, shuffle=True)\n#     return train_loader","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:16.000717Z","iopub.execute_input":"2024-01-30T09:37:16.002294Z","iopub.status.idle":"2024-01-30T09:37:16.014627Z","shell.execute_reply.started":"2024-01-30T09:37:16.002223Z","shell.execute_reply":"2024-01-30T09:37:16.011230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class MyDataset(Dataset):\n#     def __init__(self, paths):\n#         self.paths = paths\n#         self.transform = transforms.Compose([\n#             transforms.ToPILImage(),\n#             transforms.Resize((CFG.input_size, CFG.input_size)),\n#             transforms.ToTensor(),\n#             transforms.Normalize(mean=[0.5], std=[0.5])\n#         ])\n\n#     def __len__(self):\n#         return len(self.paths)\n\n#     def __getitem__(self, index):\n#         img_path = self.paths[index]\n#         img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n#         img = to_1024(img)\n#         img = self.transform(img)\n\n#         label_path = get_corresponding_label_path(img_path)\n#         label = cv2.imread(label_path, cv2.IMREAD_GRAYSCALE)\n#         label = to_1024(label)\n#         label = self.transform(label)  # Resize labels to the same size as images\n\n#         label = label.clone()\n#         label = (label > 0).to(torch.float32)\n\n#         return img, label","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:16.019673Z","iopub.execute_input":"2024-01-30T09:37:16.020823Z","iopub.status.idle":"2024-01-30T09:37:16.034688Z","shell.execute_reply.started":"2024-01-30T09:37:16.020768Z","shell.execute_reply":"2024-01-30T09:37:16.032703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_corresponding_label_path(img_path):\n#     return img_path.replace('/images/', '/labels/')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:16.037066Z","iopub.execute_input":"2024-01-30T09:37:16.038545Z","iopub.status.idle":"2024-01-30T09:37:16.053726Z","shell.execute_reply.started":"2024-01-30T09:37:16.038489Z","shell.execute_reply":"2024-01-30T09:37:16.051650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to build and load the model\n\n# class CustomModel(nn.Module):\n#     def __init__(self, CFG, weight=None):\n#         super().__init__()\n#         self.CFG = CFG\n#         self.model = smp.Unet(\n#             encoder_name=CFG.backbone,\n#             encoder_weights=weight,\n#             in_channels=CFG.in_chans,\n#             classes=CFG.target_size,\n#             activation=None,\n#         )\n#         self.batch = CFG.batch\n\n#     def forward_(self, image):\n#         output = self.model(image)\n#         return output[:, 0]\n\n#     def forward(self, x):\n#         x = x.to(torch.float32)\n#         x = norm_with_clip(x).reshape(x.shape)\n\n#         if CFG.input_size != CFG.image_size:\n#             x = nn.functional.interpolate(x.unsqueeze(1), size=(CFG.input_size, CFG.input_size), mode='bilinear', align_corners=True)\n\n#         with torch.cuda.amp.autocast():\n#             with torch.no_grad():\n#                 x = [self.forward_(x[i * self.batch:(i + 1) * self.batch].unsqueeze(1)) for i in range(x.shape[0] // self.batch + 1)]\n#                 x = torch.cat(x, dim=0)\n\n#         x = x.sigmoid()\n#         x = x.mean(dim=0)\n\n#         if CFG.input_size != CFG.image_size:\n#             x = nn.functional.interpolate(x.unsqueeze(0).unsqueeze(0), size=(CFG.image_size, CFG.image_size), mode='bilinear', align_corners=True)[0, 0]\n\n#         x = nn.functional.interpolate(x.unsqueeze(0).unsqueeze(0), size=(1024, 1024), mode='bilinear', align_corners=True)[0, 0]\n\n#         return x","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:58:47.835927Z","iopub.execute_input":"2024-01-30T10:58:47.836500Z","iopub.status.idle":"2024-01-30T10:58:47.853115Z","shell.execute_reply.started":"2024-01-30T10:58:47.836454Z","shell.execute_reply":"2024-01-30T10:58:47.851462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_paths = glob('/kaggle/input/blood-vessel-segmentation/train/**/*.tif', recursive=True)\n# test_paths = glob('/kaggle/input/blood-vessel-segmentation/test/**/*.tif', recursive=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:00.408048Z","iopub.execute_input":"2024-01-30T10:59:00.408562Z","iopub.status.idle":"2024-01-30T10:59:03.341038Z","shell.execute_reply.started":"2024-01-30T10:59:00.408524Z","shell.execute_reply":"2024-01-30T10:59:03.339689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# additional_train_folders = [\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_1_dense/labels',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_1_voi/images',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_1_voi/labels',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_2/images',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_2/labels',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_3_dense/labels',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_3_sparse/images',\n#     '/kaggle/input/blood-vessel-segmentation/train/kidney_3_sparse/labels'\n# ]\n\n# for folder in additional_train_folders:\n#     train_paths += glob(os.path.join(folder, '**/*.tif'), recursive=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:03.343276Z","iopub.execute_input":"2024-01-30T10:59:03.343741Z","iopub.status.idle":"2024-01-30T10:59:03.441249Z","shell.execute_reply.started":"2024-01-30T10:59:03.343698Z","shell.execute_reply":"2024-01-30T10:59:03.440118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and normalize data\n\n# train_data = load_and_normalize_data(train_paths)\n# test_data = load_and_normalize_data(test_paths)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:18.568610Z","iopub.execute_input":"2024-01-30T09:37:18.569000Z","iopub.status.idle":"2024-01-30T09:37:18.574684Z","shell.execute_reply.started":"2024-01-30T09:37:18.568959Z","shell.execute_reply":"2024-01-30T09:37:18.573394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_paths, valid_paths = train_test_split(train_paths, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:04.199221Z","iopub.execute_input":"2024-01-30T10:59:04.199641Z","iopub.status.idle":"2024-01-30T10:59:04.214968Z","shell.execute_reply.started":"2024-01-30T10:59:04.199606Z","shell.execute_reply":"2024-01-30T10:59:04.214040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(train_paths), len(valid_paths)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:05.476979Z","iopub.execute_input":"2024-01-30T10:59:05.477641Z","iopub.status.idle":"2024-01-30T10:59:05.484923Z","shell.execute_reply.started":"2024-01-30T10:59:05.477606Z","shell.execute_reply":"2024-01-30T10:59:05.483805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare data loaders\n\n# train_loader = prepare_data_loaders(train_paths, batch_size=16)\n\n#2\n\n# train_loader = DataLoader(MyDataset(train_paths), batch_size=CFG.batch, shuffle=True)\n# valid_loader = prepare_data_loaders(valid_paths, CFG.batch)\n# test_loader = DataLoader(MyDataset(test_paths), batch_size=1, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:07.560522Z","iopub.execute_input":"2024-01-30T10:59:07.560943Z","iopub.status.idle":"2024-01-30T10:59:07.569239Z","shell.execute_reply.started":"2024-01-30T10:59:07.560909Z","shell.execute_reply":"2024-01-30T10:59:07.568200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install torch torchvision torchaudio -f https://download.pytorch.org/whl/cu102/torch_stable.html","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:18.623986Z","iopub.execute_input":"2024-01-30T09:37:18.624408Z","iopub.status.idle":"2024-01-30T09:37:18.636713Z","shell.execute_reply.started":"2024-01-30T09:37:18.624375Z","shell.execute_reply":"2024-01-30T09:37:18.635459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and load the model\n\n# model = build_and_load_model()\n# model.load_state_dict(tc.load('/kaggle/input/SenNet_HOA_model.pth', \"cpu\"))\n# model.eval()\n# model = tc.nn.DataParallel(model)\n\n#2\n# model = build_and_load_model()\n# model.load_state_dict(tc.load('/kaggle/input/SenNet_HOA_model.pth', \"cpu\"))","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:11.804437Z","iopub.execute_input":"2024-01-30T10:59:11.804854Z","iopub.status.idle":"2024-01-30T10:59:12.852098Z","shell.execute_reply.started":"2024-01-30T10:59:11.804820Z","shell.execute_reply":"2024-01-30T10:59:12.850763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:37:19.375698Z","iopub.execute_input":"2024-01-30T09:37:19.376070Z","iopub.status.idle":"2024-01-30T09:37:19.382633Z","shell.execute_reply.started":"2024-01-30T09:37:19.376038Z","shell.execute_reply":"2024-01-30T09:37:19.380989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# criterion = nn.BCEWithLogitsLoss()\n# optimizer = optim.Adam(model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:13.590757Z","iopub.execute_input":"2024-01-30T10:59:13.591226Z","iopub.status.idle":"2024-01-30T10:59:13.598900Z","shell.execute_reply.started":"2024-01-30T10:59:13.591185Z","shell.execute_reply":"2024-01-30T10:59:13.597645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# num_epochs = 5","metadata":{"execution":{"iopub.status.busy":"2024-01-30T10:59:15.807405Z","iopub.execute_input":"2024-01-30T10:59:15.807865Z","iopub.status.idle":"2024-01-30T10:59:15.813258Z","shell.execute_reply.started":"2024-01-30T10:59:15.807826Z","shell.execute_reply":"2024-01-30T10:59:15.812182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for epoch in range(num_epochs):\n#     model.train()\n#     total_loss = 0.0\n\n#     for images, labels in tqdm(train_loader, desc=f'Epoch {epoch + 1}/{num_epochs}'):\n#         optimizer.zero_grad()\n\n#         # images, labels = images.to(device), labels.to(device)\n\n#         outputs = model(images)\n\n#         loss = criterion(outputs, labels)\n\n#         loss.backward()\n#         optimizer.step()\n\n#         total_loss += loss.item()\n\n#     average_loss = total_loss / len(train_loader)\n#     print(f'Epoch {epoch + 1}/{num_epochs}, Loss: {average_loss:.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T11:12:40.580362Z","iopub.execute_input":"2024-01-30T11:12:40.580830Z","iopub.status.idle":"2024-01-30T11:12:40.586827Z","shell.execute_reply.started":"2024-01-30T11:12:40.580792Z","shell.execute_reply":"2024-01-30T11:12:40.585311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Verifying the model on test data","metadata":{}},{"cell_type":"code","source":"# model.eval()  \n# submission_df = [] ","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:45:57.247303Z","iopub.status.idle":"2024-01-30T09:45:57.247773Z","shell.execute_reply.started":"2024-01-30T09:45:57.247559Z","shell.execute_reply":"2024-01-30T09:45:57.247581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for img, img_paths in test_loader:\n#     with torch.no_grad():\n#         img = img.to(device)\n#         output = model(img)\n\n#     rle = model_output_to_rle(output) #!!!!!!!!!!!!!!!!!\n\n#     ids = [os.path.basename(img_path).replace('.tif', '') for img_path in img_paths]\n\n#     submission_df.extend([{'id': idx, 'rle': rle[idx]} for idx in ids])","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:45:57.249218Z","iopub.status.idle":"2024-01-30T09:45:57.249612Z","shell.execute_reply.started":"2024-01-30T09:45:57.249421Z","shell.execute_reply":"2024-01-30T09:45:57.249439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df = pd.DataFrame(submission_df)\n# submission_df.to_csv('/kaggle/working/submission.csv', index=False)\n\n# print(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:45:57.250907Z","iopub.status.idle":"2024-01-30T09:45:57.251334Z","shell.execute_reply.started":"2024-01-30T09:45:57.251119Z","shell.execute_reply":"2024-01-30T09:45:57.251144Z"},"trusted":true},"execution_count":null,"outputs":[]}]}