{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Configs","metadata":{}},{"cell_type":"code","source":"class ROOTDIR:\n    train = \"./train\"\n    train_mask = \"./train_masks\"","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:25.335727Z","iopub.execute_input":"2026-09-29T13:20:25.335983Z","iopub.status.idle":"2026-09-29T13:20:25.366002Z","shell.execute_reply.started":"2026-09-29T13:20:25.335924Z","shell.execute_reply":"2026-09-29T13:20:25.365171Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# General Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport zipfile\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:25.367907Z","iopub.execute_input":"2026-09-29T13:20:25.368341Z","iopub.status.idle":"2026-09-29T13:20:25.629735Z","shell.execute_reply.started":"2026-09-29T13:20:25.368301Z","shell.execute_reply":"2026-09-29T13:20:25.62887Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Extracting Files","metadata":{}},{"cell_type":"code","source":"dirs = [\"../input/carvana-image-masking-challenge/train.zip\",\n        \"../input/carvana-image-masking-challenge/train_masks.zip\",\n        \"../input/carvana-image-masking-challenge/metadata.csv.zip\"]\n\nfor i in tqdm(dirs):\n    with zipfile.ZipFile(i) as z:\n        z.extractall()\n    ","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:25.63071Z","iopub.execute_input":"2026-09-29T13:20:25.631Z","iopub.status.idle":"2026-09-29T13:20:32.5446Z","shell.execute_reply.started":"2026-09-29T13:20:25.630976Z","shell.execute_reply":"2026-09-29T13:20:32.54376Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Working with CSV","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"./metadata.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.545597Z","iopub.execute_input":"2026-09-29T13:20:32.545873Z","iopub.status.idle":"2026-09-29T13:20:32.577931Z","shell.execute_reply.started":"2026-09-29T13:20:32.545849Z","shell.execute_reply":"2026-09-29T13:20:32.577095Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.579001Z","iopub.execute_input":"2026-09-29T13:20:32.579588Z","iopub.status.idle":"2026-09-29T13:20:32.60138Z","shell.execute_reply.started":"2026-09-29T13:20:32.579553Z","shell.execute_reply":"2026-09-29T13:20:32.600473Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Working with Train Images and Masks","metadata":{}},{"cell_type":"code","source":"train_img_lst = os.listdir(ROOTDIR.train) # \"./train\"\ntrain_mask_lst = os.listdir(ROOTDIR.train_mask) # \"./train_masks\"","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.602541Z","iopub.execute_input":"2026-09-29T13:20:32.603352Z","iopub.status.idle":"2026-09-29T13:20:32.613546Z","shell.execute_reply.started":"2026-09-29T13:20:32.603327Z","shell.execute_reply":"2026-09-29T13:20:32.612737Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_mask_lst[:5])\nprint(train_img_lst[:5])","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.615827Z","iopub.execute_input":"2026-09-29T13:20:32.616559Z","iopub.status.idle":"2026-09-29T13:20:32.620736Z","shell.execute_reply.started":"2026-09-29T13:20:32.616521Z","shell.execute_reply":"2026-09-29T13:20:32.619877Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(train_mask_lst))\nprint(len(train_img_lst))","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.621705Z","iopub.execute_input":"2026-09-29T13:20:32.621966Z","iopub.status.idle":"2026-09-29T13:20:32.633695Z","shell.execute_reply.started":"2026-09-29T13:20:32.621945Z","shell.execute_reply":"2026-09-29T13:20:32.632851Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Sorting to make sure we get right image and right mask","metadata":{}},{"cell_type":"code","source":"sorted_train_mask_lst = sorted(train_mask_lst)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.634881Z","iopub.execute_input":"2026-09-29T13:20:32.635277Z","iopub.status.idle":"2026-09-29T13:20:32.645053Z","shell.execute_reply.started":"2026-09-29T13:20:32.635242Z","shell.execute_reply":"2026-09-29T13:20:32.644175Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sorted_train_img_lst = sorted(train_img_lst)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.646039Z","iopub.execute_input":"2026-09-29T13:20:32.646429Z","iopub.status.idle":"2026-09-29T13:20:32.657903Z","shell.execute_reply.started":"2026-09-29T13:20:32.646405Z","shell.execute_reply":"2026-09-29T13:20:32.657205Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sorted_train_mask_lst[:16])\nprint(sorted_train_img_lst[:16])","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.658959Z","iopub.execute_input":"2026-09-29T13:20:32.65936Z","iopub.status.idle":"2026-09-29T13:20:32.670206Z","shell.execute_reply.started":"2026-09-29T13:20:32.659335Z","shell.execute_reply":"2026-09-29T13:20:32.669432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualizing Images with their Mask\n### Making sure images and mask are paired correctly.","metadata":{}},{"cell_type":"code","source":"def show_images(imgs_lst,masks_lst,loops=2):\n    for i in range(loops):\n        img_path = os.path.join(ROOTDIR.train,imgs_lst[i])\n        mask_path = os.path.join(ROOTDIR.train_mask,masks_lst[i])\n        img = Image.open(img_path)\n        mask = Image.open(mask_path)\n        print(img_path)\n        print(img.size)\n        print(type(img))\n        plt.imshow(img)\n        plt.show()\n        print(mask_path)\n        print(mask.size)\n        plt.imshow(mask)\n        plt.show()\n        print(\"----------------------------------------------------\")\n\nshow_images(sorted_train_img_lst, sorted_train_mask_lst)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:32.671497Z","iopub.execute_input":"2026-09-29T13:20:32.671846Z","iopub.status.idle":"2026-09-29T13:20:34.44051Z","shell.execute_reply.started":"2026-09-29T13:20:32.671813Z","shell.execute_reply":"2026-09-29T13:20:34.439657Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PyTorch Imports","metadata":{}},{"cell_type":"code","source":"import torch\nimport torchvision\nimport torch.nn as nn\nimport albumentations as A\nimport torch.optim as optim\nfrom torchvision import models\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nimport torchvision.datasets as datasets\nimport torchvision.transforms as transforms\nfrom torchvision.datasets import ImageFolder\nfrom albumentations.pytorch import ToTensorV2 \nfrom torch.utils.data import DataLoader, Dataset","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:34.4417Z","iopub.execute_input":"2026-09-29T13:20:34.442019Z","iopub.status.idle":"2026-09-29T13:20:37.521811Z","shell.execute_reply.started":"2026-09-29T13:20:34.441993Z","shell.execute_reply":"2026-09-29T13:20:37.521162Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PyTorch Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    split_pct = 0.2\n    learning_rate = 3e-4\n    batch_size = 4\n    epochs = 3","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:37.522821Z","iopub.execute_input":"2026-09-29T13:20:37.52326Z","iopub.status.idle":"2026-09-29T13:20:37.788358Z","shell.execute_reply.started":"2026-09-29T13:20:37.523234Z","shell.execute_reply":"2026-09-29T13:20:37.787403Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seed = 123\nnp.random.seed(seed)\ntorch.manual_seed(seed)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:37.789668Z","iopub.execute_input":"2026-09-29T13:20:37.789948Z","iopub.status.idle":"2026-09-29T13:20:38.079155Z","shell.execute_reply.started":"2026-09-29T13:20:37.789924Z","shell.execute_reply":"2026-09-29T13:20:38.078291Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CFG.device","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:38.080312Z","iopub.execute_input":"2026-09-29T13:20:38.080631Z","iopub.status.idle":"2026-09-29T13:20:38.086428Z","shell.execute_reply.started":"2026-09-29T13:20:38.080606Z","shell.execute_reply":"2026-09-29T13:20:38.085602Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Working with data","metadata":{}},{"cell_type":"markdown","source":"### Shuffling the data.","metadata":{}},{"cell_type":"code","source":"permuted_train_img_lst = np.random.permutation(np.array(sorted_train_img_lst))\npermuted_train_mask_lst = [x.replace(\".jpg\", \"_mask.gif\") for x in permuted_train_img_lst]\nprint(permuted_train_img_lst[:5])\nprint(permuted_train_mask_lst[:5])","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:38.087678Z","iopub.execute_input":"2026-09-29T13:20:38.088063Z","iopub.status.idle":"2026-09-29T13:20:38.105238Z","shell.execute_reply.started":"2026-09-29T13:20:38.088025Z","shell.execute_reply":"2026-09-29T13:20:38.104087Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_images(permuted_train_img_lst,permuted_train_mask_lst)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:38.106687Z","iopub.execute_input":"2026-09-29T13:20:38.107104Z","iopub.status.idle":"2026-09-29T13:20:39.90043Z","shell.execute_reply.started":"2026-09-29T13:20:38.107055Z","shell.execute_reply":"2026-09-29T13:20:39.899753Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Splitting into Training and Validation","metadata":{}},{"cell_type":"code","source":"length = len(permuted_train_img_lst)\nprint(length*0.2) # convert this to int","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:39.901729Z","iopub.execute_input":"2026-09-29T13:20:39.902127Z","iopub.status.idle":"2026-09-29T13:20:39.907327Z","shell.execute_reply.started":"2026-09-29T13:20:39.902087Z","shell.execute_reply":"2026-09-29T13:20:39.90652Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_list = permuted_train_img_lst[int(CFG.split_pct*len(permuted_train_img_lst)) :]\ntrain_masks_list = permuted_train_mask_lst[int(CFG.split_pct*len(permuted_train_mask_lst)) :]\nprint(len(train_masks_list))\n\nval_images_list = permuted_train_img_lst[: int(CFG.split_pct*len(permuted_train_img_lst))]\nval_masks_list = permuted_train_mask_lst[: int(CFG.split_pct*len(permuted_train_mask_lst))]\nprint(len(val_masks_list))\n\n# 4071+1017=5088 (split includes all items)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:39.908328Z","iopub.execute_input":"2026-09-29T13:20:39.908696Z","iopub.status.idle":"2026-09-29T13:20:39.919907Z","shell.execute_reply.started":"2026-09-29T13:20:39.908657Z","shell.execute_reply":"2026-09-29T13:20:39.919189Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualizing Train Dataset","metadata":{}},{"cell_type":"code","source":"show_images(train_images_list,train_masks_list)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:39.920927Z","iopub.execute_input":"2026-09-29T13:20:39.921239Z","iopub.status.idle":"2026-09-29T13:20:41.493395Z","shell.execute_reply.started":"2026-09-29T13:20:39.921216Z","shell.execute_reply":"2026-09-29T13:20:41.492601Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualizing Validation Dataset","metadata":{}},{"cell_type":"code","source":"show_images(val_images_list,val_masks_list)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:41.498657Z","iopub.execute_input":"2026-09-29T13:20:41.499019Z","iopub.status.idle":"2026-09-29T13:20:43.257839Z","shell.execute_reply.started":"2026-09-29T13:20:41.498992Z","shell.execute_reply":"2026-09-29T13:20:43.256992Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset Class","metadata":{}},{"cell_type":"code","source":"class CarvanaDataset(Dataset):\n    def __init__(self,img_list,mask_list,transform=None):\n        self.img_list = img_list\n        self.mask_list = mask_list\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.img_list)\n    \n    def __getitem__(self,index):\n        img_path = os.path.join(ROOTDIR.train,self.img_list[index])\n        mask_path = os.path.join(ROOTDIR.train_mask,self.mask_list[index])\n        img = Image.open(img_path)\n        mask = Image.open(mask_path)\n        img = np.array(img)\n        mask = np.array(mask)\n        \n        # TODO 1: Convert binary mask from {0, 255} to {0, 1}\n        mask[mask == 255.0] = ...\n        \n        if self.transform:\n            augmentation = self.transform(image=img, mask=mask)\n            img = augmentation[\"image\"]\n            mask = augmentation[\"mask\"]\n            \n            # TODO 2: Add channel dimension to mask\n            mask = torch.unsqueeze(mask, ...)\n            #transformations = self.transform(image=img, mask=mask)\n            #img = transformations[\"image\"]\n            #mask = transformations[\"mask\"]\n            \n        return img,mask","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.259099Z","iopub.execute_input":"2026-09-29T13:20:43.259868Z","iopub.status.idle":"2026-09-29T13:20:43.269341Z","shell.execute_reply.started":"2026-09-29T13:20:43.259828Z","shell.execute_reply":"2026-09-29T13:20:43.268231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_transform = A.Compose([\n    # TODO 3: Resize every image and mask to 572 x 572\n    A.Resize(..., ...),\n\n    A.Rotate(limit=15, p=0.1),\n\n    # TODO 4: Randomly flip images horizontally\n    A.HorizontalFlip(p=...),\n\n    A.Normalize(\n        mean=(0,0,0),\n        std=(1,1,1),\n        max_pixel_value=255\n    ),\n    ToTensorV2()\n])\n\nval_transform = A.Compose([A.Resize(572,572),\n                           A.Normalize(mean=(0,0,0),std=(1,1,1),max_pixel_value=255),\n                           ToTensorV2()])","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.270713Z","iopub.execute_input":"2026-09-29T13:20:43.271058Z","iopub.status.idle":"2026-09-29T13:20:43.311025Z","shell.execute_reply.started":"2026-09-29T13:20:43.271014Z","shell.execute_reply":"2026-09-29T13:20:43.310014Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = CarvanaDataset(train_images_list, train_masks_list, transform = train_transform)\nval_dataset = CarvanaDataset(val_images_list, val_masks_list, transform = train_transform)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.312624Z","iopub.execute_input":"2026-09-29T13:20:43.313065Z","iopub.status.idle":"2026-09-29T13:20:43.319882Z","shell.execute_reply.started":"2026-09-29T13:20:43.313016Z","shell.execute_reply":"2026-09-29T13:20:43.319061Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"idx = 200\nimg,mask = train_dataset[idx]","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.321044Z","iopub.execute_input":"2026-09-29T13:20:43.321389Z","iopub.status.idle":"2026-09-29T13:20:43.413979Z","shell.execute_reply.started":"2026-09-29T13:20:43.321338Z","shell.execute_reply":"2026-09-29T13:20:43.41299Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask.shape","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.415333Z","iopub.execute_input":"2026-09-29T13:20:43.415727Z","iopub.status.idle":"2026-09-29T13:20:43.422262Z","shell.execute_reply.started":"2026-09-29T13:20:43.415692Z","shell.execute_reply":"2026-09-29T13:20:43.421432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.max()","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.423618Z","iopub.execute_input":"2026-09-29T13:20:43.424372Z","iopub.status.idle":"2026-09-29T13:20:43.432846Z","shell.execute_reply.started":"2026-09-29T13:20:43.424322Z","shell.execute_reply":"2026-09-29T13:20:43.432161Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_single_img(img,mask,index=None,train=True):\n    if index:\n        if train:\n            img,mask = train_dataset[index]\n        else:\n            img,mask = val_dataset[index]\n    plt.imshow(img.permute(1,2,0),cmap=\"gray\")  # Convert (3, 572, 572) -> (572, 572, 3)\n    plt.show()\n    plt.imshow(mask.permute(1,2,0), cmap=\"gray\")  # Convert (1, 572, 572) -> (572, 572, 1)\n    print(mask.shape)\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.433516Z","iopub.execute_input":"2026-09-29T13:20:43.433788Z","iopub.status.idle":"2026-09-29T13:20:43.440524Z","shell.execute_reply.started":"2026-09-29T13:20:43.433763Z","shell.execute_reply":"2026-09-29T13:20:43.44002Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"---------------Train---------------\")\nshow_single_img(img,mask,index=15,train=False)\nprint(\"---------------Validation---------------\")\nshow_single_img(img,mask,index=15,train=True)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:43.441314Z","iopub.execute_input":"2026-09-29T13:20:43.443895Z","iopub.status.idle":"2026-09-29T13:20:44.424807Z","shell.execute_reply.started":"2026-09-29T13:20:43.443861Z","shell.execute_reply":"2026-09-29T13:20:44.423834Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataloader","metadata":{}},{"cell_type":"code","source":"train_dataloader = DataLoader(train_dataset,batch_size=CFG.batch_size,shuffle=True)\nval_dataloader = DataLoader(val_dataset,batch_size=CFG.batch_size,shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.425957Z","iopub.execute_input":"2026-09-29T13:20:44.426348Z","iopub.status.idle":"2026-09-29T13:20:44.432221Z","shell.execute_reply.started":"2026-09-29T13:20:44.426303Z","shell.execute_reply":"2026-09-29T13:20:44.431373Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a = iter(train_dataloader)\nimg,mask = a.next()\nprint(img.shape,mask.shape)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.433118Z","iopub.execute_input":"2026-09-29T13:20:44.433485Z","iopub.status.idle":"2026-09-29T13:20:44.680242Z","shell.execute_reply.started":"2026-09-29T13:20:44.433413Z","shell.execute_reply":"2026-09-29T13:20:44.679193Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Utility Functions","metadata":{}},{"cell_type":"code","source":"def double_conv(in_ch, out_ch):\n    conv = nn.Sequential(\n        nn.Conv2d(in_channels=in_ch,out_channels=out_ch,kernel_size=3,stride=1,padding=1),\n        nn.BatchNorm2d(out_ch),                                                            \n        nn.ReLU(inplace=True),\n        nn.Conv2d(in_channels=out_ch,out_channels=out_ch,kernel_size=3,stride=1,padding=1), \n        nn.BatchNorm2d(out_ch),                                                            \n        nn.ReLU(inplace=True)\n    )\n    \n    return conv\n\n#def cropper(og_tensor, target_tensor):\n#    og_shape = og_tensor.shape[2]\n#    target_shape = target_tensor.shape[2]\n#    delta = (og_shape - target_shape) // 2\n#    cropped_og_tensor = og_tensor[:,:,delta:og_shape-delta,delta:og_shape-delta]\n#    return cropped_og_tensor\n \n    \ndef padder(left_tensor, right_tensor): \n    # left_tensor is the tensor on the encoder side of UNET\n    # right_tensor is the tensor on the decoder side  of the UNET\n    \n    if left_tensor.shape != right_tensor.shape:\n        padded = torch.zeros(left_tensor.shape)\n        padded[:, :, :right_tensor.shape[2], :right_tensor.shape[3]] = right_tensor\n        return padded.to(CFG.device)\n    \n    return right_tensor.to(CFG.device)\n    ","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.681504Z","iopub.execute_input":"2026-09-29T13:20:44.681883Z","iopub.status.idle":"2026-09-29T13:20:44.689622Z","shell.execute_reply.started":"2026-09-29T13:20:44.681847Z","shell.execute_reply":"2026-09-29T13:20:44.688625Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# UNET MODEL FROM SCRATCH","metadata":{}},{"cell_type":"code","source":"class UNET(nn.Module):\n    def __init__(self,in_chnls, n_classes):\n        super(UNET,self).__init__()\n        \n        self.in_chnls = in_chnls\n        self.n_classes = n_classes\n\n        # TODO 5: In the encoder, reduce the height and width of the feature map by 2 times.\n        # Hint: use a 2x2 window and move it 2 pixels each step.\n        self.max_pool = nn.MaxPool2d(kernel_size= ...,stride=...)\n        \n        self.down_conv_1 = double_conv(in_ch=self.in_chnls,out_ch=64)\n        self.down_conv_2 = double_conv(in_ch=64,out_ch=128)\n        self.down_conv_3 = double_conv(in_ch=128,out_ch=256)\n        self.down_conv_4 = double_conv(in_ch=256,out_ch=512)\n        self.down_conv_5 = double_conv(in_ch=512,out_ch=1024)\n        #print(self.down_conv_1)\n\n        # TODO 6: In the decoder, increase the height and width of the feature map by 2 times.\n        self.up_conv_trans_1 = nn.ConvTranspose2d(in_channels=1024,out_channels=512,kernel_size=...,stride=...)\n        self.up_conv_trans_2 = nn.ConvTranspose2d(in_channels=512,out_channels=256,kernel_size=...,stride=...)\n        self.up_conv_trans_3 = nn.ConvTranspose2d(in_channels=256,out_channels=128,kernel_size=...,stride=...)\n        self.up_conv_trans_4 = nn.ConvTranspose2d(in_channels=128,out_channels=64,kernel_size=...,stride=...)\n        \n        self.up_conv_1 = double_conv(in_ch=1024,out_ch=512)\n        self.up_conv_2 = double_conv(in_ch=512,out_ch=256)\n        self.up_conv_3 = double_conv(in_ch=256,out_ch=128)\n        self.up_conv_4 = double_conv(in_ch=128,out_ch=64)\n\n        # TODO 8: Convert 64 feature maps into the final segmentation mask\n        self.conv_1x1 = nn.Conv2d(\n            in_channels=64,\n            out_channels=...,\n            kernel_size=1,\n            stride=1\n        )\n    def forward(self,x):\n        \n        x1 = self.down_conv_1(x)\n        #print(\"X1\", x1.shape)\n        p1 = self.max_pool(x1)\n        #print(\"p1\", p1.shape)\n        x2 = self.down_conv_2(p1)\n        #print(\"X2\", x2.shape)\n        p2 = self.max_pool(x2)\n        #print(\"p2\", p2.shape)\n        x3 = self.down_conv_3(p2)\n        #print(\"X2\", x3.shape)\n        p3 = self.max_pool(x3)\n        #print(\"p3\", p3.shape)\n        x4 = self.down_conv_4(p3)\n        #print(\"X4\", x4.shape)\n        p4 = self.max_pool(x4)\n        #print(\"p4\", p4.shape)\n        x5 = self.down_conv_5(p4)\n        #print(\"X5\", x5.shape)\n        \n        d1 = self.up_conv_trans_1(x5)  # up transpose convolution (\"up sampling\" as called in UNET paper)\n        pad1 = padder(x4,d1) # padding d1 to match x4 shape\n        \n        # TODO 9: Concatenate encoder and decoder features\n        cat1 = torch.cat([x4,pad1],dim=...) # concatenating padded d1 and x4 on channel dimension(dim 1) [batch(dim 0),channel(dim 1),height(dim 2),width(dim 3)]\n        uc1 = self.up_conv_1(cat1) # 1st up double convolution\n        \n        d2 = self.up_conv_trans_2(uc1)\n        pad2 = padder(x3,d2)\n        cat2 = torch.cat([x3,pad2],dim=...)\n        uc2 = self.up_conv_2(cat2)\n        \n        d3 = self.up_conv_trans_3(uc2)\n        pad3 = padder(x2,d3)\n        cat3 = torch.cat([x2,pad3],dim=...)\n        uc3 = self.up_conv_3(cat3)\n        \n        d4 = self.up_conv_trans_4(uc3)\n        pad4 = padder(x1,d4)\n        cat4 = torch.cat([x1,pad4],dim=...)\n        uc4 = self.up_conv_4(cat4)\n\n        conv_1x1 = self.conv_1x1(uc4)\n        return conv_1x1\n        #print(conv_1x1.shape)","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.690778Z","iopub.execute_input":"2026-09-29T13:20:44.69154Z","iopub.status.idle":"2026-09-29T13:20:44.706543Z","shell.execute_reply.started":"2026-09-29T13:20:44.691493Z","shell.execute_reply":"2026-09-29T13:20:44.705724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training and Validation","metadata":{}},{"cell_type":"markdown","source":"### Train Function\n","metadata":{}},{"cell_type":"code","source":"def train_model(model,dataloader,criterion,optimizer):\n    model.train()\n    train_running_loss = 0.0\n    for j,img_mask in enumerate(tqdm(dataloader)):\n        img = img_mask[0].float().to(CFG.device)\n        #print(\" ----- IMAGE -----\")\n        #print(img)\n        mask = img_mask[1].float().to(CFG.device)\n        #print(\" ----- MASK -----\")\n        #print(mask)\n        \n        y_pred = model(img)\n        #print(\" ----- Y PRED -----\")\n        #print(y_pred)\n        #print(\" ----- Y PRED SHAPE -----\")#\n        #print(y_pred.shape)\n        optimizer.zero_grad()\n        \n        loss = criterion(y_pred,mask)\n        \n        train_running_loss += loss.item() * CFG.batch_size\n        \n        loss.backward()\n        optimizer.step()\n        \n    train_loss = train_running_loss / (j+1)\n    return train_loss","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.707441Z","iopub.execute_input":"2026-09-29T13:20:44.707714Z","iopub.status.idle":"2026-09-29T13:20:44.722268Z","shell.execute_reply.started":"2026-09-29T13:20:44.707692Z","shell.execute_reply":"2026-09-29T13:20:44.721576Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Validation Function","metadata":{}},{"cell_type":"code","source":"def val_model(model,dataloader,criterion,optimizer):\n    model.eval()\n    val_running_loss = 0\n    with torch.no_grad():\n        for j,img_mask in enumerate(tqdm(dataloader)):\n            img = img_mask[0].float().to(CFG.device)\n            mask = img_mask[1].float().to(CFG.device)\n            y_pred = model(img)\n            loss = criterion(y_pred,mask)\n            \n            val_running_loss += loss.item() * CFG.batch_size\n            \n        val_loss = val_running_loss / (j+1)\n    return val_loss","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.723293Z","iopub.execute_input":"2026-09-29T13:20:44.723646Z","iopub.status.idle":"2026-09-29T13:20:44.733224Z","shell.execute_reply.started":"2026-09-29T13:20:44.723613Z","shell.execute_reply":"2026-09-29T13:20:44.732521Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = UNET(in_chnls = 3, n_classes = 1).to(CFG.device)\n# TODO 9: Choose a suitable loss and optimizer for binary segmentation\n\noptimizer = ...\ncriterion = ...\ntrain_loss_lst = []\nval_loss_lst = []  ","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:44.734219Z","iopub.execute_input":"2026-09-29T13:20:44.734579Z","iopub.status.idle":"2026-09-29T13:20:48.037607Z","shell.execute_reply.started":"2026-09-29T13:20:44.734544Z","shell.execute_reply":"2026-09-29T13:20:48.036909Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train and Validation Loop","metadata":{}},{"cell_type":"code","source":"for i in tqdm(range(CFG.epochs)):\n    train_loss = train_model(model=model,dataloader=train_dataloader,criterion=criterion,optimizer=optimizer)\n    val_loss = val_model(model=model,dataloader=val_dataloader,criterion=criterion,optimizer=optimizer)\n    train_loss_lst.append(train_loss)\n    val_loss_lst.append(val_loss)\n    print(f\" Train Loss : {train_loss:.4f}\")\n    print(f\" Validation Loss : {val_loss:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2026-09-29T13:20:48.038565Z","iopub.execute_input":"2026-09-29T13:20:48.038858Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Training and Validation Loss Plot","metadata":{}},{"cell_type":"code","source":"plt.plot(train_loss_lst, color=\"green\", label='train loss')\nplt.plot(val_loss_lst, color=\"red\", label='validation loss')\nplt.xlabel(\"epochs\")\nplt.ylabel(\"loss\")\nplt.legend()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Saving Model","metadata":{}},{"cell_type":"code","source":"TRAINED_FILE = \"./unet_scratch.pth\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(model.state_dict(), TRAINED_FILE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(TRAINED_FILE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Testing","metadata":{}},{"cell_type":"code","source":"trained_model = UNET(in_chnls = 3, n_classes = 1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"UNET_TRAINED = \"../input/unet-4-epoch-trained/unet_scratch.pth\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trained_model.load_state_dict(torch.load(UNET_TRAINED))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trained_model = trained_model.to(\"cuda\")\ntrained_model.eval()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_path = \"../input/carvana-image-masking-challenge/29bb3ece3180_11.jpg\"\n\nimg = cv2.imread(img_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(img)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_transform = A.Compose([A.Resize(572,572),\n                           A.Normalize(mean=(0,0,0),std=(1,1,1),max_pixel_value=255),\n                           ToTensorV2()])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image = test_transform(image = img)\n\nprint(test_image)\n\nprint(test_image[\"image\"].dtype)\nprint(test_image[\"image\"].shape)\n\nimg = test_image[\"image\"].unsqueeze(0)\nprint(img.shape)\n\nimg = img.to(\"cuda\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = trained_model(img)\npred.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask = pred.squeeze(0).cpu().detach().numpy()\nprint(mask.shape)\nmask = mask.transpose(1,2,0)\nprint(mask.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_test_img = test_image[\"image\"].cpu().detach().numpy()\nprint(display_test_img.shape)\ndisplay_test_img = display_test_img.transpose(1,2,0)\ndisplay_test_img.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask[mask < 0]=0\nmask[mask > 0]=1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"-------Original Image-------\")\nplt.imshow(display_test_img, cmap=\"gray\")\nplt.show()\nprint(\"-------Image Mask-------\")\nplt.imshow(mask,cmap=\"gray\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}