{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6927,"databundleVersionId":45059,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing the Necessary Packages","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport zipfile\nimport tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nimport torchvision\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T15:14:31.083768Z","iopub.execute_input":"2024-10-17T15:14:31.084126Z","iopub.status.idle":"2024-10-17T15:14:38.387937Z","shell.execute_reply.started":"2024-10-17T15:14:31.084088Z","shell.execute_reply":"2024-10-17T15:14:38.386946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting Started","metadata":{}},{"cell_type":"markdown","source":"### Extract the Data from Zip folders, and convert them into proper format","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/carvana-image-masking-challenge\"\nTRAIN_DIR_ZIP = os.path.join(BASE_DIR, \"train.zip\")\nTRAIN_DIR_HQ_ZIP = os.path.join(BASE_DIR, \"train_hq.zip\")\nTRAIN_MASKS_CSV = os.path.join(BASE_DIR, \"train_masks.csv.zip\") # no need of this data\nTRAIN_MASK = os.path.join(BASE_DIR, \"train_masks.zip\")\nTEST_DIR = os.path.join(BASE_DIR, \"test.zip\")\nSAMPLE_SUB = os.path.join(BASE_DIR, \"sample_submission.csv.zip\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:38.389523Z","iopub.execute_input":"2024-10-17T15:14:38.389973Z","iopub.status.idle":"2024-10-17T15:14:38.395807Z","shell.execute_reply.started":"2024-10-17T15:14:38.389938Z","shell.execute_reply":"2024-10-17T15:14:38.39482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train images\nwith zipfile.ZipFile(TRAIN_DIR_ZIP) as z:\n    z.extractall()\n    \n# HQ train images    \n# with zipfile.ZipFile(TRAIN_DIR_HQ_ZIP) as z:\n#     z.extractall()\n    \n# train masks    \nwith zipfile.ZipFile(TRAIN_MASKS_CSV) as z:\n    z.extractall()\n\n# sample submission\n# with zipfile.ZipFile(SAMPLE_SUB) as z:\n#     z.extractall()\n\n    \n# with zipfile.ZipFile(TRAIN_MASK) as z:\n#     z.extractall()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:38.396944Z","iopub.execute_input":"2024-10-17T15:14:38.397323Z","iopub.status.idle":"2024-10-17T15:14:48.859766Z","shell.execute_reply.started":"2024-10-17T15:14:38.397278Z","shell.execute_reply":"2024-10-17T15:14:48.858975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ndef ClearCudaCache():\n    torch.cuda.empty_cache()\n    print(torch.cuda.memory_summary(device=DEVICE, abbreviated=True))\nif (DEVICE.type == 'cuda'):\n    print(torch.cuda.current_device())  # This will return the index of the current device\n    print(torch.cuda.get_device_name(torch.cuda.current_device()))\nprint(f\"Device available is {DEVICE}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:48.862425Z","iopub.execute_input":"2024-10-17T15:14:48.862826Z","iopub.status.idle":"2024-10-17T15:14:48.925288Z","shell.execute_reply.started":"2024-10-17T15:14:48.862781Z","shell.execute_reply":"2024-10-17T15:14:48.924235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# paths of working directory\n\nTRAIN_DIR_BASE_PATH = \"/kaggle/working/train\"\nTRAIN_DIR = [os.path.join(TRAIN_DIR_BASE_PATH, i) for i in os.listdir(TRAIN_DIR_BASE_PATH)]\n\n# TRAIN_DIR_HQ_BASE_PATH = \"/kaggle/working/train_hq\"\n# TRAIN_DIR_HQ = [os.path.join(TRAIN_DIR_HQ_BASE_PATH, i) for i in os.listdir(TRAIN_DIR_HQ_BASE_PATH)]\n\n\n\nTRAIN_MASKS = \"/kaggle/working/train_masks\" # no need of this data\n\nTRAIN_MASKS_CSV = \"/kaggle/working/train_masks.csv\"\ntrain_mask = pd.read_csv(TRAIN_MASKS_CSV)\n\nMASK_HEIGHT, MASK_WIDTH = 1280, 1918\n\nprint(f\"total images present in the train is {len(TRAIN_DIR)}\")\n# print(f\"total images present in the train HQ is {len(TRAIN_DIR_HQ)}\")\n# print(f\"total images present in the test is {len(TEST_DIR)}\")\nprint(f\"the size of dataframe is {train_mask.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:48.927316Z","iopub.execute_input":"2024-10-17T15:14:48.927873Z","iopub.status.idle":"2024-10-17T15:14:49.405683Z","shell.execute_reply.started":"2024-10-17T15:14:48.927827Z","shell.execute_reply":"2024-10-17T15:14:49.404703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_to_mask(rle, height, width):\n    rle_numbers = list(map(int, rle.split()))\n    mask = np.zeros(height * width, dtype=np.uint8)\n\n    for i in range(0, len(rle_numbers), 2):\n        start = rle_numbers[i] - 1  # Start position (0-indexed)\n        length = rle_numbers[i + 1]  # Length of the run\n        mask[start:start + length] = 1  # Set the pixels to 1\n\n    return mask.reshape((height, width))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:49.407044Z","iopub.execute_input":"2024-10-17T15:14:49.407708Z","iopub.status.idle":"2024-10-17T15:14:49.413758Z","shell.execute_reply.started":"2024-10-17T15:14:49.407652Z","shell.execute_reply":"2024-10-17T15:14:49.412831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mask.head(), train_mask.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:49.414918Z","iopub.execute_input":"2024-10-17T15:14:49.415267Z","iopub.status.idle":"2024-10-17T15:14:49.434072Z","shell.execute_reply.started":"2024-10-17T15:14:49.415235Z","shell.execute_reply":"2024-10-17T15:14:49.433129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3):\n    tmp = TRAIN_DIR[i]\n    \n    file_name = tmp.split('/')[-1]\n    rle_mask = train_mask[train_mask['img'] == file_name]['rle_mask'].values[0]\n    mask = rle_to_mask(rle_mask, MASK_HEIGHT, MASK_WIDTH)\n\n    plt.subplot(1, 3, 1)\n    plt.imshow(cv2.imread(tmp))\n\n    tmp = TRAIN_DIR[i]\n    plt.subplot(1, 3, 2)\n    plt.imshow(cv2.imread(tmp))\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(mask)\n    \n    plt.subplots_adjust(wspace=1)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:49.435235Z","iopub.execute_input":"2024-10-17T15:14:49.435917Z","iopub.status.idle":"2024-10-17T15:14:52.158479Z","shell.execute_reply.started":"2024-10-17T15:14:49.43587Z","shell.execute_reply":"2024-10-17T15:14:52.157596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## checking the size of image and mask\n","metadata":{}},{"cell_type":"code","source":"TRAIN_DIR_BASE_PATH\nfor i in range(2):\n    idx = np.random.randint(0, 5089)\n    img = cv2.imread(os.path.join(TRAIN_DIR_BASE_PATH, train_mask.loc[idx, 'img']))\n    mask = rle_to_mask(train_mask.loc[idx, 'rle_mask'], MASK_HEIGHT, MASK_WIDTH)\n    print(img.shape, mask.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:52.159874Z","iopub.execute_input":"2024-10-17T15:14:52.160533Z","iopub.status.idle":"2024-10-17T15:14:52.190905Z","shell.execute_reply.started":"2024-10-17T15:14:52.160486Z","shell.execute_reply":"2024-10-17T15:14:52.189941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert data into proper format\n","metadata":{}},{"cell_type":"code","source":"tmp_images = []\nfor i in tqdm.tqdm(range(len(train_mask['img']))):\n    train_mask.loc[i, 'img'] = os.path.join(TRAIN_DIR_BASE_PATH, train_mask.loc[i, 'img'])\n    tmp = rle_to_mask(train_mask['rle_mask'].values[0], MASK_HEIGHT, MASK_WIDTH)\n    tmp_images.append(tmp)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:14:52.195418Z","iopub.execute_input":"2024-10-17T15:14:52.195737Z","iopub.status.idle":"2024-10-17T15:15:03.179397Z","shell.execute_reply.started":"2024-10-17T15:14:52.195703Z","shell.execute_reply":"2024-10-17T15:15:03.178505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### now train_mask dataframe contains the image path and also its mask.","metadata":{}},{"cell_type":"code","source":"train_mask['img_mask'] = tmp_images\ntrain_mask.sample(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:03.180823Z","iopub.execute_input":"2024-10-17T15:15:03.181149Z","iopub.status.idle":"2024-10-17T15:15:03.420017Z","shell.execute_reply.started":"2024-10-17T15:15:03.181115Z","shell.execute_reply":"2024-10-17T15:15:03.419044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split the data into training and validation","metadata":{}},{"cell_type":"code","source":"train_mask.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:03.421453Z","iopub.execute_input":"2024-10-17T15:15:03.422208Z","iopub.status.idle":"2024-10-17T15:15:04.001804Z","shell.execute_reply.started":"2024-10-17T15:15:03.422158Z","shell.execute_reply":"2024-10-17T15:15:04.000877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_valid, y_train, y_valid = train_test_split(train_mask['img'].values, train_mask['img_mask'].values, shuffle=True, test_size=0.25)\nprint(f\"x_train shape is {x_train.shape} | y_train shape is {y_train.shape}\")\nprint(f\"x_valid shape is {x_valid.shape} | y_valid shape is {y_valid.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.0031Z","iopub.execute_input":"2024-10-17T15:15:04.003481Z","iopub.status.idle":"2024-10-17T15:15:04.011755Z","shell.execute_reply.started":"2024-10-17T15:15:04.003436Z","shell.execute_reply":"2024-10-17T15:15:04.010838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.012877Z","iopub.execute_input":"2024-10-17T15:15:04.013628Z","iopub.status.idle":"2024-10-17T15:15:04.021221Z","shell.execute_reply.started":"2024-10-17T15:15:04.013564Z","shell.execute_reply":"2024-10-17T15:15:04.02023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create custome dataset class\n","metadata":{}},{"cell_type":"code","source":"class CustomeDataset(Dataset):\n    def __init__(self, img_path, mask_arr, transforms=None):\n        self.img_path = img_path\n        self.mask_arr = mask_arr\n        self.transforms = transforms \n        \n    def __len__(self):\n        return len(self.img_path)\n    \n    def __getitem__(self, idx):\n        img = cv2.imread(self.img_path[idx])\n#         img = img.transpose(2, 0, 1)\n        mask = self.mask_arr[idx]\n#         print(img.shape)\n        if(self.transforms):\n            img = self.transforms(img)\n            mask = self.transforms(mask)\n        \n        return img, mask","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.022511Z","iopub.execute_input":"2024-10-17T15:15:04.022832Z","iopub.status.idle":"2024-10-17T15:15:04.031837Z","shell.execute_reply.started":"2024-10-17T15:15:04.022777Z","shell.execute_reply":"2024-10-17T15:15:04.031058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = torchvision.transforms.Compose([\n    torchvision.transforms.ToPILImage(),\n    torchvision.transforms.Grayscale(num_output_channels=1),\n    torchvision.transforms.Resize((150, 150)),\n    torchvision.transforms.ToTensor()\n])\ntransform","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.03308Z","iopub.execute_input":"2024-10-17T15:15:04.03377Z","iopub.status.idle":"2024-10-17T15:15:04.044714Z","shell.execute_reply.started":"2024-10-17T15:15:04.033706Z","shell.execute_reply":"2024-10-17T15:15:04.043931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = CustomeDataset(x_train, y_train, transform)\nvalid_dataset = CustomeDataset(x_valid, y_valid, transform)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.045828Z","iopub.execute_input":"2024-10-17T15:15:04.046153Z","iopub.status.idle":"2024-10-17T15:15:04.05339Z","shell.execute_reply.started":"2024-10-17T15:15:04.046121Z","shell.execute_reply":"2024-10-17T15:15:04.05255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Select the batch_size optimaaly, because if you set it to high you will get \"cuda out of memory error\", so try and error with this parameter\n","metadata":{}},{"cell_type":"code","source":"train_dataloader = DataLoader(train_dataset, batch_size=8)\nvalid_dataloader = DataLoader(valid_dataset, batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.054464Z","iopub.execute_input":"2024-10-17T15:15:04.054877Z","iopub.status.idle":"2024-10-17T15:15:04.064033Z","shell.execute_reply.started":"2024-10-17T15:15:04.054833Z","shell.execute_reply":"2024-10-17T15:15:04.063212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_train_loader = next(iter(train_dataloader))\ntmp_valid_loader = next(iter(valid_dataloader))\nprint(tmp_train_loader[0].shape, tmp_train_loader[1].shape)\nprint(tmp_valid_loader[0].shape, tmp_valid_loader[0].shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.065058Z","iopub.execute_input":"2024-10-17T15:15:04.065365Z","iopub.status.idle":"2024-10-17T15:15:04.696305Z","shell.execute_reply.started":"2024-10-17T15:15:04.065333Z","shell.execute_reply":"2024-10-17T15:15:04.695333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create UNet Model (Method 1)","metadata":{}},{"cell_type":"markdown","source":"#### Earlier when I created a deep network with filters as [32, 64, 128, 256, 512, 1024], I'm getting CUDA out of memory error. then i decrease the number of filters. ","metadata":{}},{"cell_type":"code","source":"def double_conv(in_c, out_c):\n    tmp = nn.Sequential(\n        nn.Conv2d(in_c, out_c, kernel_size=3),\n        nn.ReLU(inplace=True),\n        nn.Conv2d(out_c, out_c, kernel_size=3),\n        nn.ReLU(inplace=True)\n    )\n    return tmp\n\ndef crop_img(tensor, target_tensor):\n    x = nn.functional.interpolate(tensor, size=target_tensor.shape[2:], mode='bilinear', align_corners=True)\n    return x","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.697542Z","iopub.execute_input":"2024-10-17T15:15:04.697866Z","iopub.status.idle":"2024-10-17T15:15:04.703975Z","shell.execute_reply.started":"2024-10-17T15:15:04.697831Z","shell.execute_reply":"2024-10-17T15:15:04.702936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Unet(nn.Module):\n    def __init__(self, in_c, out_c, originalDim=(MASK_HEIGHT, MASK_WIDTH)):\n        self.originalDim = originalDim\n        super(Unet, self).__init__()\n        self.down_conv1 = double_conv(in_c, 16)\n        self.down_conv2 = double_conv(16, 32)\n        self.down_conv3 = double_conv(32, 64)\n        self.down_conv4 = double_conv(64, 128)\n\n        self.max_pool = nn.MaxPool2d(2, 2)\n\n        self.up_trans1 = nn.ConvTranspose2d(in_channels=128, out_channels=64, kernel_size=2, stride=2)\n        self.up_conv1 = double_conv(128, 64)        \n\n        self.up_trans2 = nn.ConvTranspose2d(in_channels=64, out_channels=32, kernel_size=2, stride=2)\n        self.up_conv2 = double_conv(64, 32)    \n\n        self.up_trans3 = nn.ConvTranspose2d(in_channels=32, out_channels=16, kernel_size=2, stride=2)\n        self.up_conv3 = double_conv(32, 16)            \n        \n        self.final = nn.Conv2d(16, out_c, kernel_size=1)\n        \n    def forward(self, x):\n        # encoder \n        x1 = self.down_conv1(x) \n        x2 = self.max_pool(x1)  \n        x3 = self.down_conv2(x2) \n        x4 = self.max_pool(x3) \n        x5 = self.down_conv3(x4) \n        x6 = self.max_pool(x5)  \n        x7 = self.down_conv4(x6) \n\n        # decoder \n        x8 = self.up_trans1(x7)\n        x = torch.cat([crop_img(x5, x8), x8], 1)\n        x9 = self.up_conv1(x)\n\n        x10 = self.up_trans2(x9) \n        x = torch.cat([crop_img(x3, x10), x10], 1) \n        x11 = self.up_conv2(x)\n\n        x12 = self.up_trans3(x11)\n        x = torch.cat([crop_img(x1, x12), x12], 1) \n        x13 = self.up_conv3(x)\n        \n        x_final = self.final(x13)\n            \n        _x = nn.functional.interpolate(x_final, self.originalDim)\n        # print(f\"model ouput size is {_x.shape}\")\n        return _x\n        ","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.705184Z","iopub.execute_input":"2024-10-17T15:15:04.705481Z","iopub.status.idle":"2024-10-17T15:15:04.719128Z","shell.execute_reply.started":"2024-10-17T15:15:04.705441Z","shell.execute_reply":"2024-10-17T15:15:04.71833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ClearCudaCache() \ndef test(): \n    unet = Unet(3, 1)  \n    print(f\"original shape of mask is {MASK_HEIGHT, MASK_WIDTH}\")\n    tmp_img = torch.rand((1, 3, MASK_HEIGHT, MASK_WIDTH)) * 255.0 \n    pred_mask = unet(tmp_img)\n    \n    print(f\"predicted shape of mask is {pred_mask.shape}\")\ntest()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:04.720217Z","iopub.execute_input":"2024-10-17T15:15:04.720514Z","iopub.status.idle":"2024-10-17T15:15:08.594713Z","shell.execute_reply.started":"2024-10-17T15:15:04.720483Z","shell.execute_reply":"2024-10-17T15:15:08.593763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Model, Optimizers and Loss Function","metadata":{}},{"cell_type":"code","source":"unet = Unet(1, 1, (150, 150)).to(DEVICE)\noptimizer = torch.optim.Adam(unet.parameters(), lr=0.001)\nloss_fn = nn.BCEWithLogitsLoss()\noptimizer","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:08.596047Z","iopub.execute_input":"2024-10-17T15:15:08.596425Z","iopub.status.idle":"2024-10-17T15:15:08.741686Z","shell.execute_reply.started":"2024-10-17T15:15:08.59639Z","shell.execute_reply":"2024-10-17T15:15:08.740782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## rle encode","metadata":{}},{"cell_type":"code","source":"# def rle_encode(img):\n#     '''\n#     img: numpy array, 1 - mask, 0 - background\n#     Returns run length as string formated\n#     '''\n#     pixels = img.flatten()\n#     pixels[0] = 0\n#     pixels[-1] = 0\n#     runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n#     runs[1::2] -= runs[:-1:2]\n    \n#     return ' '.join(str(x) for x in runs)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:08.742811Z","iopub.execute_input":"2024-10-17T15:15:08.743113Z","iopub.status.idle":"2024-10-17T15:15:08.747275Z","shell.execute_reply.started":"2024-10-17T15:15:08.74308Z","shell.execute_reply":"2024-10-17T15:15:08.746408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(img):\n    '''\n    img: torch tensor, 1 - mask, 0 - background\n    Returns run length as string formatted\n    '''\n    # Ensure img is a 1D tensor\n    if img.dim() != 1:\n        img = img.flatten()\n\n    # Create a padded tensor to simulate the original logic\n    padded_img = torch.cat((torch.tensor([0], device=img.device), img, torch.tensor([0], device=img.device)))\n\n    # Find where the changes occur\n    runs = (padded_img[1:] != padded_img[:-1]).nonzero(as_tuple=True)[0] + 2\n\n    # Calculate run lengths\n    runs[1::2] -= runs[:-1:2]\n\n    # Convert to string\n    return ' '.join(str(x.item()) for x in runs)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:08.748315Z","iopub.execute_input":"2024-10-17T15:15:08.748596Z","iopub.status.idle":"2024-10-17T15:15:08.758643Z","shell.execute_reply.started":"2024-10-17T15:15:08.748565Z","shell.execute_reply":"2024-10-17T15:15:08.757803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the Model","metadata":{}},{"cell_type":"code","source":"if DEVICE.type == 'cuda':\n    ClearCudaCache()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:08.759741Z","iopub.execute_input":"2024-10-17T15:15:08.760114Z","iopub.status.idle":"2024-10-17T15:15:08.770945Z","shell.execute_reply.started":"2024-10-17T15:15:08.760071Z","shell.execute_reply":"2024-10-17T15:15:08.770058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### To solve the \"cuda out of memory error\", I have also used Mixed Precision Training while training the model. \n\n\n#### What it does is, while training the model and calculating the loss, it will change the datatype to float16 or float8, maintaining the accuracy and increasing the speed","metadata":{}},{"cell_type":"code","source":"scalar = torch.amp.GradScaler()\nscalar","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:15:08.771896Z","iopub.execute_input":"2024-10-17T15:15:08.772175Z","iopub.status.idle":"2024-10-17T15:15:08.783782Z","shell.execute_reply.started":"2024-10-17T15:15:08.772144Z","shell.execute_reply":"2024-10-17T15:15:08.783018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TOTAL_TRAIN_LOSS = []\nTOTAL_VALIDATION_LOSS = []\nEPOCHS = 10\n\nfor epoch in tqdm.tqdm(range(EPOCHS), leave=False):\n    # Training loop\n    temp_train_loss = []\n    unet.train()\n    \n    for img, mask in train_dataloader:  # Optional TQDM for train loop\n        img = img.to(DEVICE)\n        mask = mask.to(DEVICE).float()\n        # print(f\"mask shape is {mask.shape}\")\n        optimizer.zero_grad()  \n        \n        with torch.amp.autocast(device_type='cuda', dtype=torch.float16):\n            pred_mask = unet(img)  \n            # pred_mask = torch.squeeze(pred_mask, 1)\n            # print(f\"original mask {mask.shape} | predicted mask {pred_mask.shape}\")\n            loss = loss_fn(pred_mask, mask)  \n\n        del img \n        del mask\n        scalar.scale(loss).backward()\n        scalar.step(optimizer)\n        scalar.update()\n        \n        # torch.cuda.empty_cache() # remove unused space\n        temp_train_loss.append(loss.item()) \n\n    # Store average training loss\n    TOTAL_TRAIN_LOSS.append(np.mean(temp_train_loss))\n\n    # Validation loop\n    unet.eval()\n    temp_valid_loss = []\n    with torch.no_grad():\n        for img, mask in valid_dataloader:  # Optional TQDM for valid loop\n            img = img.to(DEVICE)\n            mask = mask.to(DEVICE).float()\n            \n            pred_mask = unet(img)\n            # pred_mask = torch.squeeze(pred_mask, 1)\n            \n            loss = loss_fn(pred_mask, mask) \n            temp_valid_loss.append(loss.item())  \n            # torch.cuda.empty_cache()\n            del img\n            del mask\n    TOTAL_VALIDATION_LOSS.append(np.mean(temp_valid_loss))\n\n    # Print statistics for each epoch\n    print(f'Epoch [{epoch + 1}/{EPOCHS}], '\n          f'Train Loss: {TOTAL_TRAIN_LOSS[-1]:.4f}, '\n          f'Validation Loss: {TOTAL_VALIDATION_LOSS[-1]:.4f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:19:16.894386Z","iopub.execute_input":"2024-10-17T15:19:16.894804Z","iopub.status.idle":"2024-10-17T15:24:29.058162Z","shell.execute_reply.started":"2024-10-17T15:19:16.894763Z","shell.execute_reply":"2024-10-17T15:24:29.056756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## save the model \n# torch.save(unet.state_dict(), '/kaggle/working/model_state_dict.pth')\n# torch.save(optimizer.state_dict(), '/kaggle/working/optimizer_state_dict.pth')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:31:43.216113Z","iopub.execute_input":"2024-10-17T15:31:43.216905Z","iopub.status.idle":"2024-10-17T15:31:43.220836Z","shell.execute_reply.started":"2024-10-17T15:31:43.216863Z","shell.execute_reply":"2024-10-17T15:31:43.219756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference Test images","metadata":{}},{"cell_type":"code","source":"# load the train model \n\n# unet.load_state_dict(torch.load('/kaggle/input/t1/pytorch/default/1/model_state_dict.pth'))\n# optimizer.load_state_dict(torch.load('/kaggle/input/t1/pytorch/default/1/optimizer_state_dict.pth'))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:18:58.273404Z","iopub.status.idle":"2024-10-17T15:18:58.273846Z","shell.execute_reply.started":"2024-10-17T15:18:58.273641Z","shell.execute_reply":"2024-10-17T15:18:58.273666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## TEST_DIR is the array of the paths for test images\n\n\nwith zipfile.ZipFile(TEST_DIR) as z:\n    z.extractall()\n    \nTEST_DIR_BASE_PATH = \"/kaggle/working/test\"\nTEST_DIR = [os.path.join(TEST_DIR_BASE_PATH, i) for i in os.listdir(TEST_DIR_BASE_PATH)]\nprint(len(TEST_DIR))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:31:51.740394Z","iopub.execute_input":"2024-10-17T15:31:51.740925Z","iopub.status.idle":"2024-10-17T15:34:55.51307Z","shell.execute_reply.started":"2024-10-17T15:31:51.740869Z","shell.execute_reply":"2024-10-17T15:34:55.512043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomTestDataset(Dataset):\n    def __init__(self, img_paths, transform):\n        self.img_paths = img_paths \n        self.transform = transform\n    \n    def __len__(self):\n        return len(self.img_paths)\n    \n    def __getitem__(self, idx):\n        img_path = self.img_paths[idx]\n        img_name = img_path.split('/')[-1]\n        img = cv2.imread(img_path)\n        if(transform):\n            img = self.transform(img)\n        return img_name, img","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:39:27.790644Z","iopub.execute_input":"2024-10-17T15:39:27.791075Z","iopub.status.idle":"2024-10-17T15:39:27.797865Z","shell.execute_reply.started":"2024-10-17T15:39:27.791031Z","shell.execute_reply":"2024-10-17T15:39:27.79683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### While inferencing on test dataset, it contains almost 100000 samples. so its impossible to load all the samples and work on them. \n\n#### the solution which helps me solve this issue\n    - use lower batch size\n    - use num_workers to asynchornously laod the data\n    - After certain iteration, store the temporary output and clear the memory    \n ","metadata":{}},{"cell_type":"code","source":"testDataset = CustomTestDataset(TEST_DIR, transform) \ntestLoader = DataLoader(testDataset, batch_size=4, shuffle=True, num_workers=4)\ntmp = next(iter(testLoader))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:39:33.319451Z","iopub.execute_input":"2024-10-17T15:39:33.319841Z","iopub.status.idle":"2024-10-17T15:39:34.174948Z","shell.execute_reply.started":"2024-10-17T15:39:33.319803Z","shell.execute_reply":"2024-10-17T15:39:34.173636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f\"Images on device: {imgs.device}, Model on device: {next(unet.parameters()).device}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:18:58.28154Z","iopub.status.idle":"2024-10-17T15:18:58.281893Z","shell.execute_reply.started":"2024-10-17T15:18:58.281714Z","shell.execute_reply":"2024-10-17T15:18:58.281732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(columns=['img', 'rle_mask'])\nunet.eval()\n\nTEST_MASK = []  \n\nfor batch_idx, (img_names, imgs) in enumerate(tqdm.tqdm(testLoader)):\n    imgs = imgs.to(DEVICE)\n    masks = unet(imgs)\n    for img_name, mask in zip(img_names, masks):\n        rle_encoded_mask = rle_encode(mask.cpu()) \n        TEST_MASK.append([img_name, rle_encoded_mask])\n    del imgs, masks\n    torch.cuda.empty_cache()\n#     print(torch.cuda.memory_summary(device=DEVICE))\n\n    # Save results every 5 batches\n    if batch_idx % 5 == 0:\n        tmp = pd.DataFrame(TEST_MASK, columns=['img', 'rle_mask'])\n        sub = pd.concat([sub, tmp], ignore_index=True)  \n        TEST_MASK = []  \n        \n\n# If there are remaining results in TEST_MASK after the loop\nif TEST_MASK:\n    tmp = pd.DataFrame(TEST_MASK, columns=['img', 'rle_mask'])\n    sub = pd.concat([sub, tmp], ignore_index=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T15:39:38.404941Z","iopub.execute_input":"2024-10-17T15:39:38.405861Z","iopub.status.idle":"2024-10-17T16:29:40.196183Z","shell.execute_reply.started":"2024-10-17T15:39:38.405811Z","shell.execute_reply":"2024-10-17T16:29:40.194949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:30:23.21983Z","iopub.execute_input":"2024-10-17T16:30:23.220746Z","iopub.status.idle":"2024-10-17T16:30:23.226959Z","shell.execute_reply.started":"2024-10-17T16:30:23.220697Z","shell.execute_reply":"2024-10-17T16:30:23.225849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:39:58.801645Z","iopub.execute_input":"2024-10-17T16:39:58.802534Z","iopub.status.idle":"2024-10-17T16:39:58.819544Z","shell.execute_reply.started":"2024-10-17T16:39:58.802481Z","shell.execute_reply":"2024-10-17T16:39:58.818212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(os.path.join(\"/kaggle/working/\", 'submission.csv'), index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:30:28.743429Z","iopub.execute_input":"2024-10-17T16:30:28.744195Z","iopub.status.idle":"2024-10-17T16:32:39.144952Z","shell.execute_reply.started":"2024-10-17T16:30:28.744155Z","shell.execute_reply":"2024-10-17T16:32:39.144152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom IPython.display import FileLink\n\nos.chdir('/kaggle/working')\nFileLink('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T16:33:07.302542Z","iopub.execute_input":"2024-10-17T16:33:07.303387Z","iopub.status.idle":"2024-10-17T16:33:07.3093Z","shell.execute_reply.started":"2024-10-17T16:33:07.303346Z","shell.execute_reply":"2024-10-17T16:33:07.308501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}