{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":86142,"databundleVersionId":9786425,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-12T17:07:21.672956Z","iopub.execute_input":"2024-10-12T17:07:21.673643Z","iopub.status.idle":"2024-10-12T17:07:23.547944Z","shell.execute_reply.started":"2024-10-12T17:07:21.673602Z","shell.execute_reply":"2024-10-12T17:07:23.547014Z"},"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install segmentation-models-pytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:07:24.963998Z","iopub.execute_input":"2024-10-12T17:07:24.964966Z","iopub.status.idle":"2024-10-12T17:07:36.399302Z","shell.execute_reply.started":"2024-10-12T17:07:24.964912Z","shell.execute_reply":"2024-10-12T17:07:36.398184Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Path to the dataset directories\ntrain_dir = '/kaggle/input/iitg-ai-overnight-hackathon-2024/dataset/dataset/train'\nlabels_dir = '/kaggle/input/iitg-ai-overnight-hackathon-2024/dataset/dataset/labels'\n\n# Initialize lists to store the image and corresponding json paths\nimage_paths = []\njson_paths = []\n\n# Iterate over each folder (201 to 579) in both train and labels directories\nfor folder_num in range(201, 580):  # Folder numbers from 201 to 579\n    train_folder_path = os.path.join(train_dir, str(folder_num))\n    labels_folder_path = os.path.join(labels_dir, str(folder_num))\n    \n    # Check if both the train and labels folders exist\n    if os.path.exists(train_folder_path) and os.path.exists(labels_folder_path):\n        # Get all .jpg files in the train folder\n        for img_file in os.listdir(train_folder_path):\n            if img_file.endswith('.jpg'):\n                # Base name of the file (without extension and suffix)\n                base_filename = img_file.split('_leftImg8bit')[0]\n                \n                # Construct the full image path\n                img_path = os.path.join(train_folder_path, img_file)\n                \n                # Find the corresponding .json file in the labels folder\n                json_file = base_filename + '_gtFine_polygons.json'\n                json_path = os.path.join(labels_folder_path, json_file)\n                \n                # Check if the corresponding .json file exists\n                if os.path.exists(json_path):\n                    # Append both paths to their respective lists\n                    image_paths.append(img_path)\n                    json_paths.append(json_path)\n                else:\n                    print(f\"Warning: JSON file not found for {img_file}\")\n\n# Create a DataFrame with image and json columns\ndf = pd.DataFrame({\n    'image': image_paths,\n    'json': json_paths\n})\n\n# Display the first few rows of the dataframe to check\nprint(df.head())\n\n# If you want to save the dataframe to a CSV file\ndf.to_csv('image_json_mapping.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-12T16:44:20.694909Z","iopub.execute_input":"2024-10-12T16:44:20.695858Z","iopub.status.idle":"2024-10-12T16:44:27.729954Z","shell.execute_reply.started":"2024-10-12T16:44:20.695817Z","shell.execute_reply":"2024-10-12T16:44:27.729093Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:45:00.333345Z","iopub.execute_input":"2024-10-12T16:45:00.334266Z","iopub.status.idle":"2024-10-12T16:45:00.341828Z","shell.execute_reply.started":"2024-10-12T16:45:00.334223Z","shell.execute_reply":"2024-10-12T16:45:00.340762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport numpy as np\n\n# Function to return unique labels from all JSON files in the dataframe\ndef return_unique_labels(df):\n    labels_list = []  # List to store all the labels found in the JSON files\n\n    # Iterate over each row in the dataframe, specifically the json column\n    for json_file_path in df['json']:\n        try:\n            # Open the JSON file and load its contents\n            with open(json_file_path, 'r') as f:\n                json_data = json.load(f)\n\n            # Iterate over the 'objects' in the JSON data to extract the labels\n            for obj in json_data['objects']:\n                labels_list.append(obj['label'])\n\n        except Exception as e:\n            print(f\"Error reading {json_file_path}: {e}\")\n            continue\n\n    # Get the unique labels by converting the list to a numpy array and finding unique elements\n    unique_labels = np.unique(np.array(labels_list))\n\n    return unique_labels\n\n# Example usage\nunique_labels = return_unique_labels(df)\nprint(\"Unique Labels:\", unique_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:45:09.527300Z","iopub.execute_input":"2024-10-12T16:45:09.527680Z","iopub.status.idle":"2024-10-12T16:47:37.858661Z","shell.execute_reply.started":"2024-10-12T16:45:09.527645Z","shell.execute_reply":"2024-10-12T16:47:37.857710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_clr = {\n 'animal': 10, 'autorickshaw': 20, 'bicycle': 30, 'billboard': 40, 'bridge': 50, 'building': 60, 'bus': 70,\n 'car': 80, 'caravan': 90, 'curb': 100, 'drivable fallback': 110, 'ego vehicle': 120, 'fallback background': 130,\n 'fence': 140, 'ground': 150, 'guard rail': 160, 'license plate': 170, 'motorcycle': 180, 'non-drivable fallback': 190,\n 'obs-str-bar-fallback': 200, 'out of roi': 210, 'parking': 220, 'person': 230, 'pole': 240, 'polegroup': 250,\n 'rail track': 260, 'rectification border': 270, 'rider': 280, 'road': 290, 'sidewalk': 300, 'sky': 310,\n 'traffic light': 320, 'traffic sign': 330, 'trailer': 340, 'train': 350, 'truck': 360, 'tunnel': 370,\n 'unlabeled': 380, 'vegetation': 390, 'vehicle fallback': 400, 'wall': 410\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:47:58.128132Z","iopub.execute_input":"2024-10-12T16:47:58.128841Z","iopub.status.idle":"2024-10-12T16:47:58.135776Z","shell.execute_reply.started":"2024-10-12T16:47:58.128799Z","shell.execute_reply":"2024-10-12T16:47:58.134841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_poly(file):\n    # this function will take a file name as argument\n    \n    # it will process all the objects in that file and returns\n    \n    # label: a list of labels for all the objects label[i] will have the corresponding vertices in vertexlist[i]\n    # len(label) == number of objects in the image\n    \n    # vertexlist: it should be list of list of vertices in tuple formate \n    # ex: [[(x11,y11), (x12,y12), (x13,y13) .. (x1n,y1n)]\n    #     [(x21,y21), (x22,y12), (x23,y23) .. (x2n,y2n)]\n    #      .....\n    #     [(xm1,ym1), (xm2,ym2), (xm3,ym3) .. (xmn,ymn)]]\n    # len(vertexlist) == number of objects in the image\n    \n    # * note that label[i] and vertextlist[i] are corresponds to the same object, one represents the type of the object\n    # the other represents the location\n    \n    # width of the image\n    # height of the image\n    label = []\n    vertexlist = []\n\n    with open(file,) as f:\n      json_data = json.load(f)\n      w = json_data['imgWidth']\n      h = json_data['imgHeight']\n      for obj in json_data['objects']:\n        label.append(obj['label'])\n        polygon = [tuple(p) for p in obj['polygon']]\n        vertexlist.append(polygon)\n\n    return w, h, label, vertexlist","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:48:01.380297Z","iopub.execute_input":"2024-10-12T16:48:01.381026Z","iopub.status.idle":"2024-10-12T16:48:01.388160Z","shell.execute_reply.started":"2024-10-12T16:48:01.380987Z","shell.execute_reply":"2024-10-12T16:48:01.387047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pathlib\nimport os\nimport pandas as pd\nfrom PIL import Image, ImageDraw\nimport numpy as np\n\n# Function to compute masks from JSON data and save them as image files\ndef compute_masks(data_df, label_clr, root_dir=''):\n    \"\"\"\n    Compute and save segmentation masks for each JSON file in the dataframe.\n    \n    Args:\n    data_df (DataFrame): DataFrame containing paths to JSON files in the 'json' column.\n    label_clr (dict): Dictionary mapping label names to color values for the masks.\n    root_dir (str): Directory to save the generated masks (default is current directory).\n    \n    Returns:\n    DataFrame: The updated dataframe with an additional 'mask' column containing paths to the saved masks.\n    \"\"\"\n    mask_list = []  # List to store paths to the generated mask images\n\n    # Iterate through each JSON file path in the dataframe\n    for row in data_df.json:\n        # Get image width, height, labels, and vertices from the JSON file\n        w, h, label, vertexlist = get_poly(row)\n        \n        # Create a new blank image (mask) with the same dimensions as the image\n        img = Image.new(\"RGB\", (w, h))\n        img1 = ImageDraw.Draw(img)\n\n        # Draw polygons for each object in the image using the vertices and labels\n        for i in range(len(label)):\n            if len(vertexlist[i]) > 1:  # Ensure valid polygons\n                img1.polygon(vertexlist[i], fill=label_clr.get(label[i], 0))  # Use the color from label_clr, default to 0 if not found\n        \n        # Convert the image to numpy array and extract the first channel (grayscale mask)\n        img = np.array(img)\n        im = Image.fromarray(img[:, :, 0])  # Convert the mask to a single-channel image\n\n        # Create the output directory if it doesn't exist\n        jsonpath = pathlib.PurePath(row)  # Extract the PurePath from the file path\n        output_dir = os.path.join(root_dir, 'output', jsonpath.parent.name)\n        pathlib.Path(output_dir).mkdir(parents=True, exist_ok=True)\n        \n        # Define the file path where the mask will be saved\n        maskpath = os.path.join(output_dir, jsonpath.stem + '.png')\n        mask_list.append(maskpath)\n        \n        # Save the generated mask as a PNG file\n        im.save(maskpath)\n\n    # Add the generated mask paths to the DataFrame in a new column 'mask'\n    data_df['mask'] = mask_list\n    \n    return data_df\n\n# Example Usage:\n# Assuming you have a 'json' column in your dataframe containing paths to the JSON files\n# and label_clr is a dictionary mapping labels to their respective colors\nroot_dir = '/kaggle/working'  # Adjust as needed for your environment\n\n# Assuming data_df is your dataframe with a column named 'json' that holds JSON file paths\ndata_df = compute_masks(df, label_clr, root_dir=root_dir)\n\n# Display the dataframe with the new 'mask' column\nprint(data_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:48:14.713414Z","iopub.execute_input":"2024-10-12T16:48:14.714227Z","iopub.status.idle":"2024-10-12T16:56:27.473942Z","shell.execute_reply.started":"2024-10-12T16:48:14.714189Z","shell.execute_reply":"2024-10-12T16:56:27.472877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test = train_test_split(data_df, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:58:18.687851Z","iopub.execute_input":"2024-10-12T16:58:18.688234Z","iopub.status.idle":"2024-10-12T16:58:19.325345Z","shell.execute_reply.started":"2024-10-12T16:58:18.688202Z","shell.execute_reply":"2024-10-12T16:58:19.324380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X_train.shape,\" \",X_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:58:42.450917Z","iopub.execute_input":"2024-10-12T16:58:42.452001Z","iopub.status.idle":"2024-10-12T16:58:42.456922Z","shell.execute_reply.started":"2024-10-12T16:58:42.451947Z","shell.execute_reply":"2024-10-12T16:58:42.455938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import segmentation_models_pytorch as smp\nimport torch\n\n# Define U-Net with an EfficientNet backbone (e.g., EfficientNet-B0)\nmodel = smp.Unet(\n    encoder_name=\"resnet34\",          # Backbone: ResNet34\n    encoder_weights=\"imagenet\",       # Pre-trained on ImageNet\n    in_channels=3,                    # Input channels (RGB)\n    classes=40,                       # Number of output classes (for multi-class segmentation)\n    activation=\"softmax\"              # Use softmax for multi-class segmentation\n)\n\n# Move the model to the GPU if available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = model.to(device)\n\n# Freeze the encoder if required (like encoder_freeze=True in Keras)\nfor param in model.encoder.parameters():\n    param.requires_grad = False\n\n# Print the model summary\nprint(model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:59:23.514608Z","iopub.execute_input":"2024-10-12T16:59:23.514996Z","iopub.status.idle":"2024-10-12T16:59:32.776598Z","shell.execute_reply.started":"2024-10-12T16:59:23.514959Z","shell.execute_reply":"2024-10-12T16:59:32.775571Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install albumentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T16:59:45.814848Z","iopub.execute_input":"2024-10-12T16:59:45.815263Z","iopub.status.idle":"2024-10-12T16:59:57.333441Z","shell.execute_reply.started":"2024-10-12T16:59:45.815222Z","shell.execute_reply":"2024-10-12T16:59:57.332170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.transforms as T\nfrom PIL import Image, ImageFilter\nimport random\n\nclass RandomHorizontalFlip:\n    \"\"\"Flip the image horizontally with probability p=1.\"\"\"\n    def _call_(self, img):\n        return img.transpose(Image.FLIP_LEFT_RIGHT)\n\nclass RandomVerticalFlip:\n    \"\"\"Flip the image vertically with probability p=1.\"\"\"\n    def _call_(self, img):\n        return img.transpose(Image.FLIP_TOP_BOTTOM)\n\nclass RandomEmboss:\n    \"\"\"Apply emboss effect to the image.\"\"\"\n    def _init_(self, alpha=1, strength=1):\n        self.alpha = alpha\n        self.strength = strength\n\n    def _call_(self, img):\n        # Apply the emboss filter from PIL\n        return img.filter(ImageFilter.EMBOSS)\n\nclass Sharpen:\n    \"\"\"Sharpen the image.\"\"\"\n    def _init_(self, alpha=1.0, lightness=1.5):\n        self.alpha = alpha\n        self.lightness = lightness\n\n    def _call_(self, img):\n        # Apply the sharpen filter from PIL\n        return img.filter(ImageFilter.SHARPEN)\n\n# Compose transformations\ntransform = T.Compose([\n    RandomHorizontalFlip(),\n    RandomVerticalFlip(),\n    RandomEmboss(),\n    Sharpen()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:02:42.042648Z","iopub.execute_input":"2024-10-12T17:02:42.043034Z","iopub.status.idle":"2024-10-12T17:02:42.052388Z","shell.execute_reply.started":"2024-10-12T17:02:42.042998Z","shell.execute_reply":"2024-10-12T17:02:42.051223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\nimport cv2\nimport random\n\n# Custom Dataset Class\nclass CustomDataset(Dataset):\n    CLASSES = list(range(40))  # Modify according to your data/problem\n\n    def _init_(self, df, classes, transform=None):\n        \"\"\"\n        Args:\n            df (pd.DataFrame): Dataframe containing image and mask file paths.\n            classes (list): List of class labels for segmentation.\n            transform (callable, optional): Optional transform to be applied on a sample.\n        \"\"\"\n        self.df = df\n        self.images_fps = df['image'].tolist()  # List of image file paths\n        self.masks_fps = df['mask'].tolist()    # List of mask file paths\n        self.class_values = [self.CLASSES.index(cls) for cls in classes]  # Class labels\n        self.transform = transform\n\n    def _len_(self):\n        return len(self.df)\n\n    # CustomDataset's _getitem_ method\nclass CustomDataset(Dataset):\n    CLASSES = list(range(40))  # 40 classes\n\n    def _getitem_(self, idx):\n        \"\"\"\n        Fetches the image and corresponding mask at the specified index.\n        \"\"\"\n        # Read the image and mask using OpenCV\n        image = cv2.imread(self.images_fps[idx], cv2.IMREAD_UNCHANGED)\n        mask = cv2.imread(self.masks_fps[idx], cv2.IMREAD_UNCHANGED)\n\n        # Resize the image and mask to the same shape (512, 512)\n        image = cv2.resize(image, (512, 512), interpolation=cv2.INTER_NEAREST)\n        mask = cv2.resize(mask, (512, 512), interpolation=cv2.INTER_NEAREST)\n\n        # Convert image to a tensor and add a channel dimension for grayscale images\n        image = torch.tensor(image).unsqueeze(0).float()  # Grayscale image (1, H, W)\n\n        # Clip the mask values to be in the valid range (0 to 39)\n        mask = torch.tensor(mask).unsqueeze(0).long()  # Convert to tensor and add channel dimension (1, H, W)\n        mask = torch.clamp(mask, min=0, max=39)  # Ensure values are between 0 and 39\n\n        # One-hot encode the mask\n        mask = torch.nn.functional.one_hot(mask, num_classes=41).float()  # One-hot encoding (H, W, 40)\n\n        return image, mask\n\n\n\n\n\n\n# Custom DataLoader\nclass CustomDataLoader(DataLoader):\n    def _init_(self, dataset, batch_size=1, shuffle=False):\n        \"\"\"\n        Args:\n            dataset (Dataset): Instance of the dataset class.\n            batch_size (int, optional): Size of each batch. Default is 1.\n            shuffle (bool, optional): Whether to shuffle the data. Default is False.\n        \"\"\"\n        super()._init_(dataset, batch_size=batch_size, shuffle=shuffle)\n\n    def _iter_(self):\n        for batch in super()._iter_():\n            yield batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:06:15.345687Z","iopub.execute_input":"2024-10-12T17:06:15.346605Z","iopub.status.idle":"2024-10-12T17:06:15.360027Z","shell.execute_reply.started":"2024-10-12T17:06:15.346564Z","shell.execute_reply":"2024-10-12T17:06:15.359052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\n\n# Custom Focal Loss Implementation\nclass FocalLoss(nn.Module):\n    def _init_(self, alpha=1, gamma=2, reduction='mean'):\n        super(FocalLoss, self)._init_()\n        self.alpha = alpha\n        self.gamma = gamma\n        self.reduction = reduction\n\n    def forward(self, inputs, targets):\n        BCE_loss = nn.functional.binary_cross_entropy_with_logits(inputs, targets, reduction='none')\n        pt = torch.exp(-BCE_loss)\n        F_loss = self.alpha * (1 - pt) ** self.gamma * BCE_loss\n\n        if self.reduction == 'mean':\n            return F_loss.mean()\n        elif self.reduction == 'sum':\n            return F_loss.sum()\n        else:\n            return F_loss\n\n# Dice Loss Implementation\nclass DiceLoss(nn.Module):\n    def _init_(self, smooth=1):\n        super(DiceLoss, self)._init_()\n        self.smooth = smooth\n\n    def forward(self, inputs, targets):\n        # Apply sigmoid to the inputs\n        inputs = torch.sigmoid(inputs)\n        \n        # Flatten label and prediction tensors\n        inputs = inputs.view(-1)\n        targets = targets.view(-1)\n        \n        # Compute Dice coefficient\n        intersection = (inputs * targets).sum()\n        dice = (2. * intersection + self.smooth) / (inputs.sum() + targets.sum() + self.smooth)\n        \n        return 1 - dice\n\n# Combined Focal + Dice Loss\nclass CombinedLoss(nn.Module):\n    def _init_(self, alpha=1, gamma=2, smooth=1):\n        super(CombinedLoss, self)._init_()\n        self.focal_loss = FocalLoss(alpha=alpha, gamma=gamma)\n        self.dice_loss = DiceLoss(smooth=smooth)\n\n    def forward(self, inputs, targets):\n        focal = self.focal_loss(inputs, targets)\n        dice = self.dice_loss(inputs, targets)\n        return focal + dice\n\n# IoU Metric Implementation\ndef iou_score(preds, targets, threshold=0.5, eps=1e-6):\n    preds = torch.sigmoid(preds) > threshold\n    preds = preds.float()\n    targets = targets.float()\n\n    intersection = (preds * targets).sum()\n    union = (preds + targets).sum() - intersection\n    iou = (intersection + eps) / (union + eps)\n\n    return iou\n\n# Model, optimizer, and loss function  # Replace with your model\noptimizer = optim.Adam(model.parameters(), lr=0.0001, eps=1e-8)  # Equivalent to clipvalue in TensorFlow\nloss_fn = CombinedLoss()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:06:21.284474Z","iopub.execute_input":"2024-10-12T17:06:21.285415Z","iopub.status.idle":"2024-10-12T17:06:21.301566Z","shell.execute_reply.started":"2024-10-12T17:06:21.285371Z","shell.execute_reply":"2024-10-12T17:06:21.300672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport torch\nfrom torch.utils.data import Dataset\n\nclass CustomDataset(Dataset):\n    CLASSES = list(range(40))  # 40 classes, 0 to 39\n\n    def __init__(self, df, classes=None, transform=None):\n        \"\"\"\n        Args:\n            df (pd.DataFrame): DataFrame containing image and mask file paths.\n            classes (list): List of class labels for segmentation (optional).\n            transform (callable, optional): Optional transform to be applied on a sample.\n        \"\"\"\n        self.df = df\n        self.images_fps = df['image'].tolist()  # List of image file paths\n        self.masks_fps = df['mask'].tolist()    # List of mask file paths\n        self.class_values = [self.CLASSES.index(cls) for cls in (classes or self.CLASSES)]\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        # Read the image and mask\n        image = cv2.imread(self.images_fps[idx], cv2.IMREAD_COLOR)  # Ensure it's read in color (RGB)\n        mask = cv2.imread(self.masks_fps[idx], cv2.IMREAD_UNCHANGED)\n\n        # Resize both image and mask to 512x512\n        image = cv2.resize(image, (512, 512), interpolation=cv2.INTER_NEAREST)\n        mask = cv2.resize(mask, (512, 512), interpolation=cv2.INTER_NEAREST)\n\n        # Convert image to tensor and ensure it has 3 channels (C, H, W)\n        image = torch.tensor(image).permute(2, 0, 1).float()  # Rearrange to (C, H, W)\n\n        # Convert mask to tensor and clamp values between 0 and 39\n        mask = torch.tensor(mask).long()\n        mask = torch.clamp(mask, min=0, max=39)\n\n        # One-hot encode the mask with num_classes=40 and permute to (C, H, W)\n        mask = torch.nn.functional.one_hot(mask, num_classes=40).permute(2, 0, 1).float()  # Ensure shape is (C, H, W)\n\n        return image, mask\n\n\nfrom torch.utils.data import DataLoader\n\n# Define dataset and data loader\nCLASSES = list(range(39))  # 40 classes for multi-class segmentation\n\n# Assuming X_train and X_test are your pandas DataFrames containing 'image' and 'mask' columns\ntrain_dataset = CustomDataset(X_train, classes=CLASSES)  # Use the previously defined CustomDataset class\ntest_dataset = CustomDataset(X_test, classes=CLASSES)\n\ntrain_dataloader = DataLoader(train_dataset, batch_size=2, shuffle=True)\ntest_dataloader = DataLoader(test_dataset, batch_size=2, shuffle=True)\n\n# Example to check the shape of an image and mask from the dataloader\nsample_image, sample_mask = next(iter(train_dataloader))\nprint(sample_image.shape)  # Expected: torch.Size([2, 1, 512, 512]) for grayscale images (batch_size, channels, height, width)\nprint(sample_mask.shape)   # Expected: torch.Size([2, 40, 512, 512]) for one-hot encoded masks (batch_size, num_classes, height, width)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:12:32.286125Z","iopub.execute_input":"2024-10-12T17:12:32.286864Z","iopub.status.idle":"2024-10-12T17:12:32.704439Z","shell.execute_reply.started":"2024-10-12T17:12:32.286823Z","shell.execute_reply":"2024-10-12T17:12:32.703413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.optim.lr_scheduler import ReduceLROnPlateau","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:12:36.969834Z","iopub.execute_input":"2024-10-12T17:12:36.970750Z","iopub.status.idle":"2024-10-12T17:12:36.974861Z","shell.execute_reply.started":"2024-10-12T17:12:36.970710Z","shell.execute_reply":"2024-10-12T17:12:36.973787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom tqdm import tqdm  # For progress bar\n\n# Define optimizer, scheduler, and criterion (loss function)\noptimizer = optim.Adam(model.parameters(), lr=0.0001)\nscheduler = ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=2, min_lr=0.000001, verbose=True)\ncriterion = CombinedLoss()  # Assuming you have defined this custom loss function\n\n# The rest of your training code remains the same...","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T17:12:38.551829Z","iopub.execute_input":"2024-10-12T17:12:38.552315Z","iopub.status.idle":"2024-10-12T17:12:38.561214Z","shell.execute_reply.started":"2024-10-12T17:12:38.552269Z","shell.execute_reply":"2024-10-12T17:12:38.559983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.optim as optim\nfrom tqdm import tqdm  # For progress bar\n\nimport torch\nimport torch.nn as nn\n\nclass DiceLoss(nn.Module):\n    def __init__(self, smooth=1e-6):\n        super(DiceLoss, self).__init__()\n        self.smooth = smooth\n\n    def forward(self, outputs, targets):\n        outputs = torch.sigmoid(outputs)  # Ensure the outputs are in the range (0, 1)\n        targets = targets.float()         # Ensure targets are float for calculations\n        \n        intersection = (outputs * targets).sum()\n        dice_score = (2. * intersection + self.smooth) / (outputs.sum() + targets.sum() + self.smooth)\n        return 1 - dice_score  # We return 1 - dice_score to minimize the loss\n\n\ndef save_checkpoint(model, epoch, val_loss, path):\n    \"\"\"\n    Saves the model checkpoint to the specified path.\n    \n    Args:\n        model (torch.nn.Module): The model to be saved.\n        epoch (int): The epoch number.\n        val_loss (float): The validation loss for this checkpoint.\n        path (str): The file path where to save the checkpoint.\n    \"\"\"\n    checkpoint = {\n        'epoch': epoch,\n        'model_state_dict': model.state_dict(),\n        'val_loss': val_loss,\n    }\n    torch.save(checkpoint, path)\n    print(f\"Checkpoint saved at {path}\")\n\n# Set up optimizer, scheduler, and criterion (loss function)\noptimizer = optim.Adam(model.parameters(), lr=0.0001)\nscheduler = ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=2, min_lr=0.000001, verbose=True)\ncriterion = DiceLoss()  # Use the custom loss function (focal + dice loss)\n\n# Move model to the appropriate device (GPU or CPU)\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n# Training function\ndef train_one_epoch(model, dataloader, optimizer, criterion, device):\n    model.train()  # Set the model to training mode\n    running_loss = 0.0\n    for inputs, targets in tqdm(dataloader, desc=\"Training\", leave=False):\n        inputs, targets = inputs.to(device), targets.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n    \n    epoch_loss = running_loss / len(dataloader)\n    return epoch_loss\n\n# Validation function\ndef validate_one_epoch(model, dataloader, criterion, device):\n    model.eval()  # Set the model to evaluation mode\n    running_loss = 0.0\n    iou_score_sum = 0.0\n    with torch.no_grad():\n        for inputs, targets in tqdm(dataloader, desc=\"Validation\", leave=False):\n            inputs, targets = inputs.to(device), targets.to(device)\n            outputs = model(inputs)\n            loss = criterion(outputs, targets)\n            \n            running_loss += loss.item()\n            iou_score_sum += iou_score(outputs, targets).item()\n    \n    epoch_loss = running_loss / len(dataloader)\n    avg_iou_score = iou_score_sum / len(dataloader)\n    return epoch_loss, avg_iou_score\n\n# Training and validation loop\nnum_epochs = 5\nbest_val_loss = float(\"inf\")\n\n# Define directory for saving checkpoints\nlog_dir = \"./checkpoints\"  # Specify your preferred directory here\nos.makedirs(log_dir, exist_ok=True)  # Create the directory if it doesn't exist\n\nfor epoch in range(num_epochs):\n    print(f\"Epoch {epoch+1}/{num_epochs}\")\n    \n    # Training step\n    train_loss = train_one_epoch(model, train_dataloader, optimizer, criterion, device)\n    print(f\"Dice Loss: {train_loss:.4f}\")\n    \n    # Validation step\n    val_loss, val_iou = validate_one_epoch(model, test_dataloader, criterion, device)\n    print(f\"Validation Loss: {val_loss:.4f}, Validation IoU: {val_iou:.4f}\")\n    \n    # Step the learning rate scheduler\n    scheduler.step(val_loss)\n    \n    # Save the best model\n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        save_checkpoint(model, epoch, val_loss, os.path.join(log_dir, 'best_model.pth'))\n        print(\"Saved Best Model\")\n\n    print(\"-\" * 30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T18:17:35.891768Z","iopub.execute_input":"2024-10-12T18:17:35.892168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\nimport cv2\nimport random\n\n# Custom Dataset Class\nclass CustomDataset(Dataset):\n    CLASSES = list(range(21))  # Modify according to your data/problem\n\n    def _init_(self, df, classes, transform=None):\n        \"\"\"\n        Args:\n            df (pd.DataFrame): Dataframe containing image and mask file paths.\n            classes (list): List of class labels for segmentation.\n            transform (callable, optional): Optional transform to be applied on a sample.\n        \"\"\"\n        self.df = df\n        self.images_fps = df['image'].tolist()  # List of image file paths\n        self.masks_fps = df['mask'].tolist()    # List of mask file paths\n        self.class_values = [self.CLASSES.index(cls) for cls in classes]  # Class labels\n        self.transform = transform\n\n    def _len_(self):\n        return len(self.df)\n\n    # CustomDataset's _getitem_ method\nclass CustomDataset(Dataset):\n    CLASSES = list(range(40))  # 40 classes\n\n    def _getitem_(self, idx):\n        \"\"\"\n        Fetches the image and corresponding mask at the specified index.\n        \"\"\"\n        # Read the image and mask using OpenCV\n        image = cv2.imread(self.images_fps[idx], cv2.IMREAD_UNCHANGED)\n        mask = cv2.imread(self.masks_fps[idx], cv2.IMREAD_UNCHANGED)\n\n        # Resize the image and mask to the same shape (512, 512)\n        image = cv2.resize(image, (512, 512), interpolation=cv2.INTER_NEAREST)\n        mask = cv2.resize(mask, (512, 512), interpolation=cv2.INTER_NEAREST)\n\n        # Convert image to a tensor and add a channel dimension for grayscale images\n        image = torch.tensor(image).unsqueeze(0).float()  # Grayscale image (1, H, W)\n\n        # Clip the mask values to be in the valid range (0 to 39)\n        mask = torch.tensor(mask).unsqueeze(0).long()  # Convert to tensor and add channel dimension (1, H, W)\n        mask = torch.clamp(mask, min=0, max=39)  # Ensure values are between 0 and 39\n\n        # One-hot encode the mask\n        mask = torch.nn.functional.one_hot(mask, num_classes=41).float()  # One-hot encoding (H, W, 40)\n\n        return image, mask\n\n\n\n\n\n\n# Custom DataLoader\nclass CustomDataLoader(DataLoader):\n    def _init_(self, dataset, batch_size=1, shuffle=False):\n        \"\"\"\n        Args:\n            dataset (Dataset): Instance of the dataset class.\n            batch_size (int, optional): Size of each batch. Default is 1.\n            shuffle (bool, optional): Whether to shuffle the data. Default is False.\n        \"\"\"\n        super()._init_(dataset, batch_size=batch_size, shuffle=shuffle)\n\n    def _iter_(self):\n        for batch in super()._iter_():\n            yield batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T18:14:30.885235Z","iopub.execute_input":"2024-10-12T18:14:30.886144Z","iopub.status.idle":"2024-10-12T18:14:30.899391Z","shell.execute_reply.started":"2024-10-12T18:14:30.886100Z","shell.execute_reply":"2024-10-12T18:14:30.898538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the entire model\ntorch.save(model, \"model_1.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T18:15:00.601998Z","iopub.execute_input":"2024-10-12T18:15:00.602408Z","iopub.status.idle":"2024-10-12T18:15:00.778025Z","shell.execute_reply.started":"2024-10-12T18:15:00.602368Z","shell.execute_reply":"2024-10-12T18:15:00.777001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot training & validation iou_score values\nplt.figure(figsize=(30, 5))\n\n# IoU Score\nplt.subplot(121)\nplt.plot(train_iou_scores, label='Train IoU Score')\nplt.plot(val_iou_scores, label='Validation IoU Score')\nplt.title('Model IoU Score')\nplt.ylabel('IoU Score')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\n# Loss\nplt.subplot(122)\nplt.plot(train_losses, label='Train Loss')\nplt.plot(val_losses, label='Validation Loss')\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(loc='upper left')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T18:15:10.399593Z","iopub.execute_input":"2024-10-12T18:15:10.400716Z","iopub.status.idle":"2024-10-12T18:15:10.713435Z","shell.execute_reply.started":"2024-10-12T18:15:10.400671Z","shell.execute_reply":"2024-10-12T18:15:10.712148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport torch\n\n# Assuming model is your trained PyTorch model\nmodel.eval()  # Set the model to evaluation mode\n\nfor idx, row in enumerate(X_test[:3].values):\n    # Original image\n    image = cv2.imread(row[0], cv2.IMREAD_UNCHANGED) \n    image = cv2.resize(image, (512, 512), interpolation=cv2.INTER_AREA)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    # Prepare the image for model input\n    image_tensor = torch.tensor(image).permute(2, 0, 1).float()  # Change to (C, H, W)\n    image_tensor = image_tensor.unsqueeze(0).to(device)  # Add batch dimension and move to device\n\n    # Predicted segmentation map\n    with torch.no_grad():  # No need to track gradients during inference\n        predicted_mask = model(image_tensor)\n    predicted_mask = torch.argmax(predicted_mask, dim=1).cpu().numpy()  # Get predicted classes and convert to numpy\n\n    # Original segmentation map\n    image_mask = cv2.imread(row[2], cv2.IMREAD_UNCHANGED)\n    image_mask = cv2.resize(image_mask, (512, 512))\n\n    # Plotting\n    plt.figure(figsize=(15, 5))\n    plt.subplot(131)\n    plt.imshow(image)\n    plt.title(\"Original Image\")\n    plt.axis('off')\n\n    plt.subplot(132)\n    plt.imshow(image_mask, cmap='gray')\n    plt.title(\"Original Segmentation Map\")\n    plt.axis('off')\n\n    plt.subplot(133)\n    plt.imshow(predicted_mask[0], cmap='gray')  # Use predicted_mask[0] to access the first image in the batch\n    plt.title(\"Predicted Segmentation Map\")\n    plt.axis('off')\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-12T18:15:21.905485Z","iopub.execute_input":"2024-10-12T18:15:21.905900Z","iopub.status.idle":"2024-10-12T18:15:23.467128Z","shell.execute_reply.started":"2024-10-12T18:15:21.905864Z","shell.execute_reply":"2024-10-12T18:15:23.466167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}