{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92860,"databundleVersionId":11074257,"sourceType":"competition"}],"dockerImageVersionId":30887,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside  of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-21T09:17:49.152534Z","iopub.execute_input":"2025-02-21T09:17:49.152789Z","iopub.status.idle":"2025-02-21T09:17:50.142399Z","shell.execute_reply.started":"2025-02-21T09:17:49.152749Z","shell.execute_reply":"2025-02-21T09:17:50.141361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hi, I am Sushant Goyal, and this is my project... don't forget to read stuff written on comments and in Markdown cells.\n\n# So, let's begin.","metadata":{}},{"cell_type":"markdown","source":"# Accessing the dataset, storing image and mask files in a sorted manner.","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\nimport random\nimport numpy as np\n\n# dataset paths for training and testing\ntrain_images_path = \"/kaggle/input/slicee-my-face/images/train\"\ntrain_masks_path  = \"/kaggle/input/slicee-my-face/annotations/train\"\n\ntest_images_path = \"/kaggle/input/slicee-my-face/images/test\"\ntest_masks_path  = \"/kaggle/input/slicee-my-face/annotations/test\"\n\nval_images_path = \"/kaggle/input/slicee-my-face/images/val\"\nval_masks_path = \"/kaggle/input/slicee-my-face/annotations/val\"\n\ntrain_image_files = sorted([\n    f for f in os.listdir(train_images_path) \n    if f.lower().endswith(('.jpg', '.jpeg', '.png'))\n])\n\ntrain_mask_files = sorted([\n    f for f in os.listdir(train_masks_path) \n    if f.lower().endswith(('.jpg', '.jpeg', '.png'))\n])\n\nprint(f\"Total train images used: {len(train_image_files)}\")\nprint(f\"Total train masks used: {len(train_mask_files)}\")\n\n\ntest_image_files = sorted([\n    f for f in os.listdir(test_images_path) \n    if f.lower().endswith(('.jpg', '.jpeg', '.png'))\n])\n\ntest_mask_files = sorted([\n    f for f in os.listdir(test_masks_path) \n    if f.lower().endswith(('.jpg', '.jpeg', '.png'))\n])\n\nprint(f\"Total test images used: {len(test_image_files)}\")\nprint(f\"Total test masks used: {len(test_mask_files)}\")\n\nval_image_files = sorted([\n    f for f in os.listdir(val_images_path) \n    if f.lower().endswith(('.jpg', '.jpeg', '.png'))\n])\n\nval_mask_files = sorted([\n    f for f in os.listdir(val_masks_path) \n    if f.lower().endswith(('.jpg', '.jpeg', '.png'))\n])\n\nprint(f\"Total val images used: {len(val_image_files)}\")\nprint(f\"Total val masks used: {len(val_mask_files)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-21T09:17:50.143411Z","iopub.execute_input":"2025-02-21T09:17:50.143888Z","iopub.status.idle":"2025-02-21T09:17:51.358732Z","shell.execute_reply.started":"2025-02-21T09:17:50.143823Z","shell.execute_reply":"2025-02-21T09:17:51.357288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing the dataset (Resizing, Normalizing, decimal -> int conversion of labels).\n# Converted data to Numpy arrays.","metadata":{}},{"cell_type":"code","source":"\"\"\"\nI wrote this script to preprocess my dataset for the segmentation task. \nIt loads images and their corresponding masks from the specified directories, resizes them to 256x256, and normalizes the image pixel values to the range [0,1]. \nSince the masks are originally grayscale with decimal values (multiples of 1/255), I convert them to integer labels by multiplying by 255, rounding the result, and remapping the unique values to a contiguous range starting from 0. \nThis script processes the training, test, and validation data separately, converts them into NumPy arrays, and prints the shapes of the processed arrays.\n\"\"\"\n\nIMG_SIZE = (256, 256)\n\ndef load_and_preprocess(image_path, mask_path):\n    \"\"\"\n    Loads an image and its mask, resizes to IMG_SIZE,\n    and normalizes pixel values to [0,1].\n    \"\"\"\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = cv2.resize(image, IMG_SIZE)\n    image = image.astype(np.float32) / 255.0\n    \n    # read mask (grayscale)\n    mask = cv2.imread(mask_path, cv2.IMREAD_GRAYSCALE)\n    mask = cv2.resize(mask, IMG_SIZE)\n    mask = mask.astype(np.float32) / 255.0\n    \n    return image, mask\n\n\n#!!!!! converting labels of mask from decimal to int\ndef convert_mask_to_int(mask):\n    \"\"\"\n    Convert a normalized mask (values in [0,1] multiples of 1/255) into an integer mask.\n    The function here multiplies by 255, rounds the result, and then remaps the unique values\n    to a contiguous range starting from 0.\n    \"\"\"\n  \n    mask_int = np.round(mask * 255).astype(np.uint8)\n    unique_vals = np.unique(mask_int)\n    mapping = {old_val: new_val for new_val, old_val in enumerate(np.sort(unique_vals))}\n    \n    remapped_mask = np.copy(mask_int)\n    for old_val, new_val in mapping.items():\n        remapped_mask[mask_int == old_val] = new_val\n        \n    return remapped_mask\n\n# process training data\nprocessed_train_images = []\nprocessed_train_masks = []\n\nfor img_file, msk_file in zip(train_image_files, train_mask_files):\n    img_path = os.path.join(train_images_path, img_file)\n    mask_path = os.path.join(train_masks_path, msk_file)\n    \n    image, mask = load_and_preprocess(img_path, mask_path)\n    mask = convert_mask_to_int(mask)  # converts mask to integer labels\n    \n    processed_train_images.append(image)\n    processed_train_masks.append(mask)\n\nprint(f\"Preprocessed {len(processed_train_images)} training images and {len(processed_train_masks)} training masks.\")\n\nprocessed_train_images_np = np.array(processed_train_images)  \nprocessed_train_masks_np  = np.array(processed_train_masks)    \n\nprint(\"processed_train_images_np shape:\", processed_train_images_np.shape)\nprint(\"processed_train_masks_np shape:\", processed_train_masks_np.shape)\n\n# process Test Data\nprocessed_test_images = []\nprocessed_test_masks = []\n\nfor img_file, msk_file in zip(test_image_files, test_mask_files):\n    img_path = os.path.join(test_images_path, img_file)\n    mask_path = os.path.join(test_masks_path, msk_file)\n    \n    image, mask = load_and_preprocess(img_path, mask_path)\n    mask = convert_mask_to_int(mask)\n    \n    processed_test_images.append(image)\n    processed_test_masks.append(mask)\n\nprint(f\"Preprocessed {len(processed_test_images)} test images and {len(processed_test_masks)} test masks.\")\n\nprocessed_test_images_np = np.array(processed_test_images)  \nprocessed_test_masks_np  = np.array(processed_test_masks)    \n\nprint(\"processed_test_images_np shape:\", processed_test_images_np.shape)\nprint(\"processed_test_masks_np shape:\", processed_test_masks_np.shape)\n\n# process val data\nprocessed_val_images = []\nprocessed_val_masks = []\n\nfor img_file, msk_file in zip(val_image_files, val_mask_files):\n    img_path = os.path.join(val_images_path, img_file)\n    mask_path = os.path.join(val_masks_path, msk_file)\n    \n    image, mask = load_and_preprocess(img_path, mask_path)\n    mask = convert_mask_to_int(mask)\n    \n    processed_val_images.append(image)\n    processed_val_masks.append(mask)\n\nprint(f\"Preprocessed {len(processed_val_images)} val images and {len(processed_val_masks)} val masks.\")\n\nprocessed_val_images_np = np.array(processed_val_images)  \nprocessed_val_masks_np  = np.array(processed_val_masks)    \n\nprint(\"processed_val_images_np shape:\", processed_val_images_np.shape)\nprint(\"processed_val_masks_np shape:\", processed_val_masks_np.shape)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-21T09:17:51.360117Z","iopub.execute_input":"2025-02-21T09:17:51.360461Z","iopub.status.idle":"2025-02-21T09:23:21.603416Z","shell.execute_reply.started":"2025-02-21T09:17:51.360429Z","shell.execute_reply":"2025-02-21T09:23:21.601706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Printing Sample Images of Preprocessed Images and Masks.","metadata":{}},{"cell_type":"code","source":"def show_random_samples(images, masks, num_samples=3):\n    for _ in range(num_samples):\n        idx = random.randint(0, len(images) - 1)\n        fig, ax = plt.subplots(1, 2, figsize=(10, 5))\n        ax[0].imshow(images[idx])\n        ax[0].set_title(\"Processed Image\")\n        ax[0].axis(\"off\")\n        ax[1].imshow(masks[idx], cmap=\"gray\")\n        ax[1].set_title(\"Processed Mask\")\n        ax[1].axis(\"off\")\n        plt.show()\n\n# display 3 random samples\nshow_random_samples(processed_train_images, processed_train_masks, num_samples=3)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nfor i in range(5):\n    unique_vals = np.unique(processed_train_masks_np[i])\n    print(f\"Mask {i} unique values:\", unique_vals)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Multi - Class Segmentation Pipeline (using SegFace model - An attention based U-net model)\n# Used the entire training dataset to train the model.\n# Trained using cross - entropy loss and Adam Optimizer.\n# Evaluated performance using Dice Coefficient.\n# Visualized predictions on Validation dataset.","metadata":{}},{"cell_type":"code","source":"\"\"\"\nI have written this script to build a multi-class segmentation pipeline using PyTorch. \nIn this code, I created a custom dataset class that handles image-mask pairs, where the images are normalized \nand the masks are in integer format representing different classes. I then built an attention-based U-Net model, \nnamed SegFace, which consists of an encoder, a bottleneck, and a decoder with attention gates to improve feature fusion. \n\nFor training, I use the entire training dataset provided (processed_train_images_np and processed_train_masks_np) \nand a separate validation dataset (processed_val_images_np and processed_val_masks_np). The model is trained using \ncross-entropy loss and the Adam optimizer, and I evaluate its performance using the Dice coefficient. The best model \n(based on validation Dice) is saved for future use. \n\nI have also included a function to visualize the predictions on the validation set so that I can easily check how well \nthe model is performing on unseen data. This script is written in a clear and understandable way, reflecting my own \napproach to solving the segmentation task.\n\"\"\"\n\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport matplotlib.pyplot as plt\n\nclass MulticlassSegDataset(Dataset):\n    def __init__(self, images_np, masks_np, transform=None):\n        self.images_np = images_np\n        self.masks_np = masks_np\n        self.transform = transform\n    def __len__(self):\n        return len(self.images_np)\n    def __getitem__(self, idx):\n        image = self.images_np[idx].astype(np.float32)\n        mask = self.masks_np[idx].astype(np.int64)\n        image = np.transpose(image, (2, 0, 1))\n        image_tensor = torch.from_numpy(image)\n        mask_tensor = torch.from_numpy(mask)\n        return image_tensor, mask_tensor\n\nX_train = processed_train_images_np\ny_train = processed_train_masks_np\nX_val = processed_val_images_np\ny_val = processed_val_masks_np\n\nprint(\"Train set images shape:\", X_train.shape, \"masks shape:\", y_train.shape)\nprint(\"Val set images shape:\", X_val.shape, \"masks shape:\", y_val.shape)\n\ntrain_dataset = MulticlassSegDataset(X_train, y_train)\nval_dataset = MulticlassSegDataset(X_val, y_val)\n\ntrain_loader = DataLoader(train_dataset, batch_size=8, shuffle=True, num_workers=2)\nval_loader = DataLoader(val_dataset, batch_size=8, shuffle=False, num_workers=2)\n\nimport os\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport matplotlib.pyplot as plt\nimport random\n\nclass MulticlassSegDataset(Dataset):\n    def __init__(self, images_np, masks_np, transform=None):\n        self.images_np = images_np\n        self.masks_np = masks_np\n        self.transform = transform\n    def __len__(self):\n        return len(self.images_np)\n    def __getitem__(self, idx):\n        image = self.images_np[idx].astype(np.float32)\n        mask = self.masks_np[idx].astype(np.int64)\n        image = np.transpose(image, (2, 0, 1))\n        image_tensor = torch.from_numpy(image)\n        mask_tensor = torch.from_numpy(mask)\n        return image_tensor, mask_tensor\n\nclass AttentionBlock(nn.Module):\n    def __init__(self, F_g, F_l, F_int):\n        super(AttentionBlock, self).__init__()\n        self.W_g = nn.Sequential(\n            nn.Conv2d(F_g, F_int, kernel_size=1, stride=1, padding=0, bias=True),\n            nn.BatchNorm2d(F_int)\n        )\n        self.W_x = nn.Sequential(\n            nn.Conv2d(F_l, F_int, kernel_size=1, stride=1, padding=0, bias=True),\n            nn.BatchNorm2d(F_int)\n        )\n        self.psi = nn.Sequential(\n            nn.Conv2d(F_int, 1, kernel_size=1, stride=1, padding=0, bias=True),\n            nn.BatchNorm2d(1),\n            nn.Sigmoid()\n        )\n        self.relu = nn.ReLU(inplace=True)\n    def forward(self, g, x):\n        g1 = self.W_g(g)\n        x1 = self.W_x(x)\n        psi = self.relu(g1 + x1)\n        psi = self.psi(psi)\n        return x * psi\n\nclass SegFace(nn.Module):\n    def __init__(self, in_channels=3, out_channels=4, features=[64, 128, 256, 512]):\n        super(SegFace, self).__init__()\n        self.encoder = nn.ModuleList()\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2)\n        for feature in features:\n            self.encoder.append(self._block(in_channels, feature))\n            in_channels = feature\n        self.bottleneck = self._block(features[-1], features[-1]*2)\n        self.decoder = nn.ModuleList()\n        self.attention_gates = nn.ModuleList()\n        for feature in reversed(features):\n            self.decoder.append(\n                nn.ConvTranspose2d(feature*2, feature, kernel_size=2, stride=2)\n            )\n            self.attention_gates.append(AttentionBlock(F_g=feature, F_l=feature, F_int=feature//2))\n            self.decoder.append(self._block(feature*2, feature))\n        self.final_conv = nn.Conv2d(features[0], out_channels, kernel_size=1)\n    def forward(self, x):\n        skip_connections = []\n        for enc in self.encoder:\n            x = enc(x)\n            skip_connections.append(x)\n            x = self.pool(x)\n        x = self.bottleneck(x)\n        skip_connections = skip_connections[::-1]\n        for idx in range(0, len(self.decoder), 2):\n            x = self.decoder[idx](x)\n            skip = skip_connections[idx//2]\n            attn_skip = self.attention_gates[idx//2](g=x, x=skip)\n            x = torch.cat((attn_skip, x), dim=1)\n            x = self.decoder[idx+1](x)\n        return F.softmax(self.final_conv(x), dim=1)\n    def _block(self, in_channels, out_channels):\n        return nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, kernel_size=3, padding=1),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_channels, out_channels, kernel_size=3, padding=1),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True)\n        )\n\ndef multiclass_dice_score(pred, target, smooth=1e-5):\n    num_classes = pred.shape[1]\n    target_onehot = F.one_hot(target, num_classes=num_classes).permute(0, 3, 1, 2).float()\n    pred_argmax = pred.argmax(dim=1, keepdim=True)\n    pred_onehot = F.one_hot(pred_argmax.squeeze(1), num_classes=num_classes).permute(0, 3, 1, 2).float()\n    intersection = (pred_onehot * target_onehot).sum(dim=(2,3))\n    union = pred_onehot.sum(dim=(2,3)) + target_onehot.sum(dim=(2,3))\n    dice_per_class = (2. * intersection + smooth) / (union + smooth)\n    return dice_per_class.mean()\n\nprint(\"Unique labels in training masks:\", np.unique(processed_train_masks_np))\nnum_classes = int(np.unique(processed_train_masks_np).max()) + 1\nprint(\"Number of classes:\", num_classes)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = SegFace(in_channels=3, out_channels=num_classes).to(device)\nprint(\"SegFace model initialized with\", num_classes, \"classes.\")\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\n\nnum_epochs = 40\nbest_val_dice = 0.0\nfor epoch in range(num_epochs):\n    model.train()\n    running_train_loss = 0.0\n    for images, masks in train_loader:\n        images, masks = images.to(device), masks.to(device)\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, masks)\n        loss.backward()\n        optimizer.step()\n        running_train_loss += loss.item() * images.size(0)\n    epoch_train_loss = running_train_loss / len(train_loader.dataset)\n    model.eval()\n    running_val_loss = 0.0\n    running_val_dice = 0.0\n    total_val_samples = 0\n    with torch.no_grad():\n        for images, masks in val_loader:\n            images, masks = images.to(device), masks.to(device)\n            outputs = model(images)\n            val_loss = criterion(outputs, masks)\n            running_val_loss += val_loss.item() * images.size(0)\n            batch_dice = multiclass_dice_score(outputs, masks)\n            running_val_dice += batch_dice.item() * images.size(0)\n            total_val_samples += images.size(0)\n    epoch_val_loss = running_val_loss / len(val_loader.dataset)\n    epoch_val_dice = running_val_dice / total_val_samples\n    print(f\"Epoch [{epoch+1}/{num_epochs}] | Train Loss: {epoch_train_loss:.4f} | Val Loss: {epoch_val_loss:.4f} | Val Dice: {epoch_val_dice:.4f}\")\n    if epoch_val_dice > best_val_dice:\n        best_val_dice = epoch_val_dice\n        torch.save(model.state_dict(), \"segface_multiclass_best.pth\")\n        print(\"** Model Saved! **\")\nprint(f\"Training complete. Best Validation Dice: {best_val_dice:.4f}\")\n\ndef visualize_predictions(dataset, model, num_samples=10):\n    indices = random.sample(range(len(dataset)), num_samples)\n    for idx in indices:\n        image_tensor, true_mask_tensor = dataset[idx]\n        image_np = np.transpose(image_tensor.numpy(), (1, 2, 0))\n        true_mask_np = true_mask_tensor.numpy()\n        image_input = image_tensor.unsqueeze(0).to(device)\n        with torch.no_grad():\n            outputs = model(image_input)\n        pred_mask = outputs.argmax(dim=1).squeeze(0).cpu().numpy()\n        fig, ax = plt.subplots(1,3, figsize=(12,4))\n        ax[0].imshow(image_np)\n        ax[0].set_title(\"Original Image\")\n        ax[0].axis(\"off\")\n        ax[1].imshow(true_mask_np, cmap=\"tab20\")\n        ax[1].set_title(\"True Mask\")\n        ax[1].axis(\"off\")\n        ax[2].imshow(pred_mask, cmap=\"tab20\")\n        ax[2].set_title(\"Predicted Mask\")\n        ax[2].axis(\"off\")\n        plt.tight_layout()\n        plt.show()\n\nvisualize_predictions(val_dataset, model, num_samples=10)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-21T10:12:58.984816Z","iopub.execute_input":"2025-02-21T10:12:58.985530Z","iopub.status.idle":"2025-02-21T10:13:06.057495Z","shell.execute_reply.started":"2025-02-21T10:12:58.985487Z","shell.execute_reply":"2025-02-21T10:13:06.055994Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualizing predictions made by model, after converting masks to a greyscale image by stretching values to (0, 255) range.\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\nThis code is for visualizing the model's predictions on random test images. \nThe model takes preprocessed test images as input and generates predicted segmentation masks. \nSince the output mask contains class labels, I convert it to a grayscale image by stretching the values across the 0-255 range. \nFor better comparison, I do the same for the true mask. \nEach visualization consists of the original test image, its corresponding ground truth mask, and the predicted mask, all displayed side by side.\n\"\"\" \n\nimport torch\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport random\n\ndef stretch_mask(mask):\n    max_val = mask.max()\n    if max_val == 0:\n        return mask.astype(np.uint8)\n    stretched = (mask.astype(float) / max_val) * 255.0\n    return stretched.astype(np.uint8)\n\n# vistualize masks of test images\ndef visualize_test_predictions(model, processed_test_images_np, processed_test_masks_np, device, num_samples=10):\n    model.eval()\n    indices = random.sample(range(processed_test_images_np.shape[0]), num_samples)\n    \n    for idx in indices:\n        image_np = processed_test_images_np[idx]\n        true_mask_np = processed_test_masks_np[idx]\n        \n        image_tensor = torch.from_numpy(np.transpose(image_np, (2, 0, 1))).float().unsqueeze(0).to(device)\n        \n        with torch.no_grad():\n            outputs = model(image_tensor)\n        pred_mask = outputs.argmax(dim=1).squeeze(0).cpu().numpy()\n        \n        pred_mask_gray = stretch_mask(pred_mask)\n        true_mask_gray = stretch_mask(true_mask_np)\n        \n        fig, ax = plt.subplots(1, 3, figsize=(15,5))\n        ax[0].imshow(image_np)\n        ax[0].set_title(f\"Test Image {idx}\")\n        ax[0].axis(\"off\")\n        \n        ax[1].imshow(true_mask_gray, cmap='gray', vmin=0, vmax=255)\n        ax[1].set_title(\"True Mask (Grayscale)\")\n        ax[1].axis(\"off\")\n        \n        ax[2].imshow(pred_mask_gray, cmap='gray', vmin=0, vmax=255)\n        ax[2].set_title(\"Predicted Mask (Grayscale)\")\n        ax[2].axis(\"off\")\n        \n        plt.tight_layout()\n        plt.show()\n\nvisualize_test_predictions(model, processed_test_images_np, processed_test_masks_np, device, num_samples=10)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-21T10:13:06.058150Z","iopub.status.idle":"2025-02-21T10:13:06.058519Z","shell.execute_reply":"2025-02-21T10:13:06.058365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Generates submission file using trained Segface model.\n# Test images are passed through model and the predictions are converted to binary masks (for submission).\n# RLE encoding is used to store the masks (masks are restored to original size before encoding).\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\nThis code is for generating the submission file using the trained SegFace model. \nThe test images are already processed, so I just pass them through the model to get predictions. \nSince it's a multi-class segmentation model, the output is converted to a binary mask by keeping all non-zero predictions as foreground. \nThe original test image size needs to be restored before encoding, so I read the image dimensions and resize the predicted mask accordingly. \nRun-length encoding (RLE) is used to store the segmentation mask in a format suitable for submission. \nFinally, the results are written to a CSV file with image IDs and their corresponding RLE-encoded masks, the sorting of masks is done alphanumerically here.\n\"\"\"\n\nimport os\nimport numpy as np\nimport torch\nimport csv\nimport cv2\n\ndef rle_encode(mask):\n    pixels = mask.flatten()  \n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    if len(runs) % 2 == 1:\n        runs = runs[:-1]\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n# generate binary masks (resize masks to original dimensions as well)\ndef generate_submission(model, processed_test_images_np, test_image_files, test_images_path, device, num_classes, submission_filename=\"submission.csv\"):\n    test_image_files = sorted(test_image_files, key=lambda x: x.lower())\n    model.eval()\n    submission_data = []\n    N = processed_test_images_np.shape[0]\n    for i in range(N):\n        image_np = processed_test_images_np[i]\n        image_tensor = torch.from_numpy(np.transpose(image_np, (2, 0, 1))).float().unsqueeze(0).to(device)\n        with torch.no_grad():\n            outputs = model(image_tensor)\n        pred_mask = outputs.argmax(dim=1).squeeze(0).cpu().numpy()\n        binary_mask = (pred_mask > 0).astype(np.uint8)\n        image_filename = test_image_files[i]\n        original_image_path = os.path.join(test_images_path, image_filename)\n        orig_img = cv2.imread(original_image_path)\n        if orig_img is None:\n            orig_h, orig_w = binary_mask.shape\n        else:\n            orig_h, orig_w = orig_img.shape[:2]\n        upscaled_mask = cv2.resize(binary_mask, (orig_w, orig_h), interpolation=cv2.INTER_NEAREST)\n        rle = rle_encode(upscaled_mask)\n        image_id = os.path.splitext(image_filename)[0]\n        submission_data.append([image_id, rle])\n    with open(submission_filename, 'w', newline='') as f:\n        writer = csv.writer(f)\n        writer.writerow([\"id\", \"predicted\"])\n        writer.writerows(submission_data)\n    print(f\"Submission file '{submission_filename}' created with {len(submission_data)} entries.\")\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.load_state_dict(torch.load(\"segface_multiclass_best.pth\", map_location=device))\nmodel.eval()\ngenerate_submission(model, processed_test_images_np, test_image_files, test_images_path, device, num_classes, submission_filename=\"submission.csv\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# The best score for this task is 0.98547 (officially ranked 11th in Kaggle Leaderboards).\n","metadata":{}},{"cell_type":"markdown","source":"# Research papers I read for this project -\n# 1) Paper on SegFace ->  https://arxiv.org/abs/2412.08647\n# 2) Paper on U - Net ->  https://arxiv.org/abs/1505.04597\n# 3) Paper on Transformer based segmentation ->  https://arxiv.org/abs/2105.05633\n# I also referred to the book named \"Hands on Machine Learning by Aurelion Geron\".","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}