{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6927,"databundleVersionId":45059,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing the Necessary Packages","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport zipfile\nimport tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nimport torchvision\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-16T06:10:55.579317Z","iopub.execute_input":"2024-10-16T06:10:55.579722Z","iopub.status.idle":"2024-10-16T06:11:04.505399Z","shell.execute_reply.started":"2024-10-16T06:10:55.579683Z","shell.execute_reply":"2024-10-16T06:11:04.504231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting Started","metadata":{}},{"cell_type":"markdown","source":"### Extract the Data from Zip folders, and convert them into proper format","metadata":{}},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/carvana-image-masking-challenge\"\nTRAIN_DIR_ZIP = os.path.join(BASE_DIR, \"train.zip\")\nTRAIN_DIR_HQ_ZIP = os.path.join(BASE_DIR, \"train_hq.zip\")\nTRAIN_MASKS_CSV = os.path.join(BASE_DIR, \"train_masks.csv.zip\") # no need of this data\nTRAIN_MASK = os.path.join(BASE_DIR, \"train_masks.zip\")\nTEST_DIR = os.path.join(BASE_DIR, \"test.zip\")\nSAMPLE_SUB = os.path.join(BASE_DIR, \"sample_submission.csv.zip\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:46:45.624734Z","iopub.execute_input":"2024-10-15T06:46:45.625296Z","iopub.status.idle":"2024-10-15T06:46:45.631801Z","shell.execute_reply.started":"2024-10-15T06:46:45.625258Z","shell.execute_reply":"2024-10-15T06:46:45.630789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train images\nwith zipfile.ZipFile(TRAIN_DIR_ZIP) as z:\n    z.extractall()\n    \n# HQ train images    \n# with zipfile.ZipFile(TRAIN_DIR_HQ_ZIP) as z:\n#     z.extractall()\n    \n# train masks    \nwith zipfile.ZipFile(TRAIN_MASKS_CSV) as z:\n    z.extractall()\n\n# sample submission\n# with zipfile.ZipFile(SAMPLE_SUB) as z:\n#     z.extractall()\n\n    \n# with zipfile.ZipFile(TRAIN_MASK) as z:\n#     z.extractall()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:46:54.563717Z","iopub.execute_input":"2024-10-15T06:46:54.564578Z","iopub.status.idle":"2024-10-15T06:47:04.007483Z","shell.execute_reply.started":"2024-10-15T06:46:54.564539Z","shell.execute_reply":"2024-10-15T06:47:04.006617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ndef ClearCudaCache():\n    torch.cuda.empty_cache()\n    print(torch.cuda.memory_summary(device=DEVICE, abbreviated=True))\nif (DEVICE.type == 'cuda'):\n    print(torch.cuda.current_device())  # This will return the index of the current device\n    print(torch.cuda.get_device_name(torch.cuda.current_device()))\nprint(f\"Device available is {DEVICE}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:06.523129Z","iopub.execute_input":"2024-10-15T06:47:06.523562Z","iopub.status.idle":"2024-10-15T06:47:06.588535Z","shell.execute_reply.started":"2024-10-15T06:47:06.523518Z","shell.execute_reply":"2024-10-15T06:47:06.587579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# paths of working directory\n\nTRAIN_DIR_BASE_PATH = \"/kaggle/working/train\"\nTRAIN_DIR = [os.path.join(TRAIN_DIR_BASE_PATH, i) for i in os.listdir(TRAIN_DIR_BASE_PATH)]\n\n# TRAIN_DIR_HQ_BASE_PATH = \"/kaggle/working/train_hq\"\n# TRAIN_DIR_HQ = [os.path.join(TRAIN_DIR_HQ_BASE_PATH, i) for i in os.listdir(TRAIN_DIR_HQ_BASE_PATH)]\n\n\n\nTRAIN_MASKS = \"/kaggle/working/train_masks\" # no need of this data\n\nTRAIN_MASKS_CSV = \"/kaggle/working/train_masks.csv\"\ntrain_mask = pd.read_csv(TRAIN_MASKS_CSV)\n\nMASK_HEIGHT, MASK_WIDTH = 1280, 1918\n\nprint(f\"total images present in the train is {len(TRAIN_DIR)}\")\n# print(f\"total images present in the train HQ is {len(TRAIN_DIR_HQ)}\")\n# print(f\"total images present in the test is {len(TEST_DIR)}\")\nprint(f\"the size of dataframe is {train_mask.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:11.984601Z","iopub.execute_input":"2024-10-15T06:47:11.985517Z","iopub.status.idle":"2024-10-15T06:47:12.459960Z","shell.execute_reply.started":"2024-10-15T06:47:11.985466Z","shell.execute_reply":"2024-10-15T06:47:12.458796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_to_mask(rle, height, width):\n    rle_numbers = list(map(int, rle.split()))\n    mask = np.zeros(height * width, dtype=np.uint8)\n\n    for i in range(0, len(rle_numbers), 2):\n        start = rle_numbers[i] - 1  # Start position (0-indexed)\n        length = rle_numbers[i + 1]  # Length of the run\n        mask[start:start + length] = 1  # Set the pixels to 1\n\n    return mask.reshape((height, width))","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:15.241779Z","iopub.execute_input":"2024-10-15T06:47:15.242189Z","iopub.status.idle":"2024-10-15T06:47:15.248951Z","shell.execute_reply.started":"2024-10-15T06:47:15.242152Z","shell.execute_reply":"2024-10-15T06:47:15.247794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mask.head(), train_mask.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:16.347470Z","iopub.execute_input":"2024-10-15T06:47:16.348184Z","iopub.status.idle":"2024-10-15T06:47:16.361997Z","shell.execute_reply.started":"2024-10-15T06:47:16.348145Z","shell.execute_reply":"2024-10-15T06:47:16.360785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3):\n    tmp = TRAIN_DIR[i]\n    \n    file_name = tmp.split('/')[-1]\n    rle_mask = train_mask[train_mask['img'] == file_name]['rle_mask'].values[0]\n    mask = rle_to_mask(rle_mask, MASK_HEIGHT, MASK_WIDTH)\n\n    plt.subplot(1, 3, 1)\n    plt.imshow(cv2.imread(tmp))\n\n    tmp = TRAIN_DIR[i]\n    plt.subplot(1, 3, 2)\n    plt.imshow(cv2.imread(tmp))\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(mask)\n    \n    plt.subplots_adjust(wspace=1)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:17.549007Z","iopub.execute_input":"2024-10-15T06:47:17.549661Z","iopub.status.idle":"2024-10-15T06:47:20.399735Z","shell.execute_reply.started":"2024-10-15T06:47:17.549619Z","shell.execute_reply":"2024-10-15T06:47:20.398715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## checking the size of image and mask\n","metadata":{}},{"cell_type":"code","source":"TRAIN_DIR_BASE_PATH\nfor i in range(2):\n    idx = np.random.randint(0, 5089)\n    img = cv2.imread(os.path.join(TRAIN_DIR_BASE_PATH, train_mask.loc[idx, 'img']))\n    mask = rle_to_mask(train_mask.loc[idx, 'rle_mask'], MASK_HEIGHT, MASK_WIDTH)\n    print(img.shape, mask.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:20.401641Z","iopub.execute_input":"2024-10-15T06:47:20.402014Z","iopub.status.idle":"2024-10-15T06:47:20.434745Z","shell.execute_reply.started":"2024-10-15T06:47:20.401981Z","shell.execute_reply":"2024-10-15T06:47:20.433736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert data into proper format\n","metadata":{}},{"cell_type":"code","source":"tmp_images = []\nfor i in tqdm.tqdm(range(len(train_mask['img']))):\n    train_mask.loc[i, 'img'] = os.path.join(TRAIN_DIR_BASE_PATH, train_mask.loc[i, 'img'])\n    tmp = rle_to_mask(train_mask['rle_mask'].values[0], MASK_HEIGHT, MASK_WIDTH)\n    tmp_images.append(tmp)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:21.704536Z","iopub.execute_input":"2024-10-15T06:47:21.705559Z","iopub.status.idle":"2024-10-15T06:47:32.681677Z","shell.execute_reply.started":"2024-10-15T06:47:21.705505Z","shell.execute_reply":"2024-10-15T06:47:32.680710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### now train_mask dataframe contains the image path and also its mask.","metadata":{}},{"cell_type":"code","source":"train_mask['img_mask'] = tmp_images\ntrain_mask.sample(2)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:38.263884Z","iopub.execute_input":"2024-10-15T06:47:38.264790Z","iopub.status.idle":"2024-10-15T06:47:38.525472Z","shell.execute_reply.started":"2024-10-15T06:47:38.264748Z","shell.execute_reply":"2024-10-15T06:47:38.524550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split the data into training and validation","metadata":{}},{"cell_type":"code","source":"train_mask.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:41.040972Z","iopub.execute_input":"2024-10-15T06:47:41.041911Z","iopub.status.idle":"2024-10-15T06:47:41.614264Z","shell.execute_reply.started":"2024-10-15T06:47:41.041866Z","shell.execute_reply":"2024-10-15T06:47:41.613307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_valid, y_train, y_valid = train_test_split(train_mask['img'].values, train_mask['img_mask'].values, shuffle=True, test_size=0.25)\nprint(f\"x_train shape is {x_train.shape} | y_train shape is {y_train.shape}\")\nprint(f\"x_valid shape is {x_valid.shape} | y_valid shape is {y_valid.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:42.725132Z","iopub.execute_input":"2024-10-15T06:47:42.725556Z","iopub.status.idle":"2024-10-15T06:47:42.734840Z","shell.execute_reply.started":"2024-10-15T06:47:42.725491Z","shell.execute_reply":"2024-10-15T06:47:42.733710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:48.786877Z","iopub.execute_input":"2024-10-15T06:47:48.787722Z","iopub.status.idle":"2024-10-15T06:47:48.793940Z","shell.execute_reply.started":"2024-10-15T06:47:48.787681Z","shell.execute_reply":"2024-10-15T06:47:48.792843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create custome dataset class\n","metadata":{}},{"cell_type":"code","source":"class CustomeDataset(Dataset):\n    def __init__(self, img_path, mask_arr, transforms=None):\n        self.img_path = img_path\n        self.mask_arr = mask_arr\n        self.transforms = transforms \n        \n    def __len__(self):\n        return len(self.img_path)\n    \n    def __getitem__(self, idx):\n        img = cv2.imread(self.img_path[idx])\n#         img = img.transpose(2, 0, 1)\n        mask = self.mask_arr[idx]\n#         print(img.shape)\n        if(self.transforms):\n            img = self.transforms(img)\n            mask = self.transforms(mask)\n        \n        return img, mask","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:52.164745Z","iopub.execute_input":"2024-10-15T06:47:52.165147Z","iopub.status.idle":"2024-10-15T06:47:52.172235Z","shell.execute_reply.started":"2024-10-15T06:47:52.165110Z","shell.execute_reply":"2024-10-15T06:47:52.171071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = torchvision.transforms.Compose([\n    torchvision.transforms.ToPILImage(),\n    torchvision.transforms.Grayscale(num_output_channels=1),\n    torchvision.transforms.Resize((150, 150)),\n    torchvision.transforms.ToTensor()\n])\ntransform","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:53.751716Z","iopub.execute_input":"2024-10-15T06:47:53.752582Z","iopub.status.idle":"2024-10-15T06:47:53.760733Z","shell.execute_reply.started":"2024-10-15T06:47:53.752533Z","shell.execute_reply":"2024-10-15T06:47:53.759557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = CustomeDataset(x_train, y_train, transform)\nvalid_dataset = CustomeDataset(x_valid, y_valid, transform)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:54.623180Z","iopub.execute_input":"2024-10-15T06:47:54.623796Z","iopub.status.idle":"2024-10-15T06:47:54.629038Z","shell.execute_reply.started":"2024-10-15T06:47:54.623747Z","shell.execute_reply":"2024-10-15T06:47:54.627973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Select the batch_size optimaaly, because if you set it to high you will get \"cuda out of memory error\", so try and error with this parameter\n","metadata":{}},{"cell_type":"code","source":"train_dataloader = DataLoader(train_dataset, batch_size=8)\nvalid_dataloader = DataLoader(valid_dataset, batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:55.872427Z","iopub.execute_input":"2024-10-15T06:47:55.873347Z","iopub.status.idle":"2024-10-15T06:47:55.878289Z","shell.execute_reply.started":"2024-10-15T06:47:55.873306Z","shell.execute_reply":"2024-10-15T06:47:55.877148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_train_loader = next(iter(train_dataloader))\ntmp_valid_loader = next(iter(valid_dataloader))\nprint(tmp_train_loader[0].shape, tmp_train_loader[1].shape)\nprint(tmp_valid_loader[0].shape, tmp_valid_loader[0].shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:56.905867Z","iopub.execute_input":"2024-10-15T06:47:56.906288Z","iopub.status.idle":"2024-10-15T06:47:57.547675Z","shell.execute_reply.started":"2024-10-15T06:47:56.906231Z","shell.execute_reply":"2024-10-15T06:47:57.546666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create UNet Model (Method 1)","metadata":{}},{"cell_type":"markdown","source":"#### Earlier when I created a deep network with filters as [32, 64, 128, 256, 512, 1024], I'm getting CUDA out of memory error. then i decrease the number of filters. ","metadata":{}},{"cell_type":"code","source":"def double_conv(in_c, out_c):\n    tmp = nn.Sequential(\n        nn.Conv2d(in_c, out_c, kernel_size=3),\n        nn.ReLU(inplace=True),\n        nn.Conv2d(out_c, out_c, kernel_size=3),\n        nn.ReLU(inplace=True)\n    )\n    return tmp\n\ndef crop_img(tensor, target_tensor):\n    x = nn.functional.interpolate(tensor, size=target_tensor.shape[2:], mode='bilinear', align_corners=True)\n    return x","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:58.184628Z","iopub.execute_input":"2024-10-15T06:47:58.185374Z","iopub.status.idle":"2024-10-15T06:47:58.191137Z","shell.execute_reply.started":"2024-10-15T06:47:58.185335Z","shell.execute_reply":"2024-10-15T06:47:58.190214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Unet(nn.Module):\n    def __init__(self, in_c, out_c, originalDim=(MASK_HEIGHT, MASK_WIDTH)):\n        self.originalDim = originalDim\n        super(Unet, self).__init__()\n        self.down_conv1 = double_conv(in_c, 16)\n        self.down_conv2 = double_conv(16, 32)\n        self.down_conv3 = double_conv(32, 64)\n        self.down_conv4 = double_conv(64, 128)\n\n        self.max_pool = nn.MaxPool2d(2, 2)\n\n        self.up_trans1 = nn.ConvTranspose2d(in_channels=128, out_channels=64, kernel_size=2, stride=2)\n        self.up_conv1 = double_conv(128, 64)        \n\n        self.up_trans2 = nn.ConvTranspose2d(in_channels=64, out_channels=32, kernel_size=2, stride=2)\n        self.up_conv2 = double_conv(64, 32)    \n\n        self.up_trans3 = nn.ConvTranspose2d(in_channels=32, out_channels=16, kernel_size=2, stride=2)\n        self.up_conv3 = double_conv(32, 16)            \n        \n        self.final = nn.Conv2d(16, out_c, kernel_size=1)\n        \n    def forward(self, x):\n        # encoder \n        x1 = self.down_conv1(x) \n        x2 = self.max_pool(x1)  \n        x3 = self.down_conv2(x2) \n        x4 = self.max_pool(x3) \n        x5 = self.down_conv3(x4) \n        x6 = self.max_pool(x5)  \n        x7 = self.down_conv4(x6) \n\n        # decoder \n        x8 = self.up_trans1(x7)\n        x = torch.cat([crop_img(x5, x8), x8], 1)\n        x9 = self.up_conv1(x)\n\n        x10 = self.up_trans2(x9) \n        x = torch.cat([crop_img(x3, x10), x10], 1) \n        x11 = self.up_conv2(x)\n\n        x12 = self.up_trans3(x11)\n        x = torch.cat([crop_img(x1, x12), x12], 1) \n        x13 = self.up_conv3(x)\n        \n        x_final = self.final(x13)\n            \n        _x = nn.functional.interpolate(x_final, self.originalDim)\n        # print(f\"model ouput size is {_x.shape}\")\n        return _x\n        ","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:58.673903Z","iopub.execute_input":"2024-10-15T06:47:58.674588Z","iopub.status.idle":"2024-10-15T06:47:58.687194Z","shell.execute_reply.started":"2024-10-15T06:47:58.674545Z","shell.execute_reply":"2024-10-15T06:47:58.686175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ClearCudaCache() \ndef test(): \n    unet = Unet(3, 1)  \n    print(f\"original shape of mask is {MASK_HEIGHT, MASK_WIDTH}\")\n    tmp_img = torch.rand((1, 3, MASK_HEIGHT, MASK_WIDTH)) * 255.0 \n    pred_mask = unet(tmp_img)\n    \n    print(f\"predicted shape of mask is {pred_mask.shape}\")\ntest()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:47:59.140192Z","iopub.execute_input":"2024-10-15T06:47:59.140875Z","iopub.status.idle":"2024-10-15T06:48:03.227335Z","shell.execute_reply.started":"2024-10-15T06:47:59.140834Z","shell.execute_reply":"2024-10-15T06:48:03.226281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Model, Optimizers and Loss Function","metadata":{}},{"cell_type":"code","source":"unet = Unet(1, 1, (150, 150)).to(DEVICE)\noptimizer = torch.optim.Adam(unet.parameters(), lr=0.001)\nloss_fn = nn.BCEWithLogitsLoss()\noptimizer","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:23:17.876054Z","iopub.execute_input":"2024-10-15T07:23:17.876453Z","iopub.status.idle":"2024-10-15T07:23:17.895609Z","shell.execute_reply.started":"2024-10-15T07:23:17.876413Z","shell.execute_reply":"2024-10-15T07:23:17.894710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## rle encode","metadata":{}},{"cell_type":"code","source":"# def rle_encode(img):\n#     '''\n#     img: numpy array, 1 - mask, 0 - background\n#     Returns run length as string formated\n#     '''\n#     pixels = img.flatten()\n#     pixels[0] = 0\n#     pixels[-1] = 0\n#     runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n#     runs[1::2] -= runs[:-1:2]\n    \n#     return ' '.join(str(x) for x in runs)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:48:14.447780Z","iopub.execute_input":"2024-10-15T06:48:14.448168Z","iopub.status.idle":"2024-10-15T06:48:14.454752Z","shell.execute_reply.started":"2024-10-15T06:48:14.448132Z","shell.execute_reply":"2024-10-15T06:48:14.453707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(img):\n    '''\n    img: torch tensor, 1 - mask, 0 - background\n    Returns run length as string formatted\n    '''\n    # Ensure img is a 1D tensor\n    if img.dim() != 1:\n        img = img.flatten()\n\n    # Create a padded tensor to simulate the original logic\n    padded_img = torch.cat((torch.tensor([0], device=img.device), img, torch.tensor([0], device=img.device)))\n\n    # Find where the changes occur\n    runs = (padded_img[1:] != padded_img[:-1]).nonzero(as_tuple=True)[0] + 2\n\n    # Calculate run lengths\n    runs[1::2] -= runs[:-1:2]\n\n    # Convert to string\n    return ' '.join(str(x.item()) for x in runs)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:23:10.440755Z","iopub.execute_input":"2024-10-15T07:23:10.441836Z","iopub.status.idle":"2024-10-15T07:23:10.451304Z","shell.execute_reply.started":"2024-10-15T07:23:10.441786Z","shell.execute_reply":"2024-10-15T07:23:10.450225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the Model","metadata":{}},{"cell_type":"code","source":"if DEVICE.type == 'cuda':\n    ClearCudaCache()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:23:11.496122Z","iopub.execute_input":"2024-10-15T07:23:11.496549Z","iopub.status.idle":"2024-10-15T07:23:11.502875Z","shell.execute_reply.started":"2024-10-15T07:23:11.496484Z","shell.execute_reply":"2024-10-15T07:23:11.501797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### To solve the \"cuda out of memory error\", I have also used Mixed Precision Training while training the model. \n\n\n#### What it does is, while training the model and calculating the loss, it will change the datatype to float16 or float8, maintaining the accuracy and increasing the speed","metadata":{}},{"cell_type":"code","source":"scalar = torch.amp.GradScaler()\nscalar","metadata":{"execution":{"iopub.status.busy":"2024-10-16T06:22:19.459849Z","iopub.execute_input":"2024-10-16T06:22:19.460992Z","iopub.status.idle":"2024-10-16T06:22:19.498878Z","shell.execute_reply.started":"2024-10-16T06:22:19.460942Z","shell.execute_reply":"2024-10-16T06:22:19.497615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TOTAL_TRAIN_LOSS = []\nTOTAL_VALIDATION_LOSS = []\nEPOCHS = 10\n\nfor epoch in tqdm.tqdm(range(EPOCHS), leave=False):\n    # Training loop\n    temp_train_loss = []\n    unet.train()\n    \n    for img, mask in train_dataloader:  # Optional TQDM for train loop\n        img = img.to(DEVICE)\n        mask = mask.to(DEVICE).float()\n        # print(f\"mask shape is {mask.shape}\")\n        optimizer.zero_grad()  \n        \n        with torch.amp.autocast(device_type='cuda', dtype=torch.float16):\n            pred_mask = unet(img)  \n            # pred_mask = torch.squeeze(pred_mask, 1)\n            # print(f\"original mask {mask.shape} | predicted mask {pred_mask.shape}\")\n            loss = loss_fn(pred_mask, mask)  \n\n        del img \n        del mask\n        scalar.scale(loss).backward()\n        scalar.step(optimizer)\n        scalar.update()\n        \n        # torch.cuda.empty_cache() # remove unused space\n        temp_train_loss.append(loss.item()) \n\n    # Store average training loss\n    TOTAL_TRAIN_LOSS.append(np.mean(temp_train_loss))\n\n    # Validation loop\n    unet.eval()\n    temp_valid_loss = []\n    with torch.no_grad():\n        for img, mask in valid_dataloader:  # Optional TQDM for valid loop\n            img = img.to(DEVICE)\n            mask = mask.to(DEVICE).float()\n            \n            pred_mask = unet(img)\n            # pred_mask = torch.squeeze(pred_mask, 1)\n            \n            loss = loss_fn(pred_mask, mask) \n            temp_valid_loss.append(loss.item())  \n            # torch.cuda.empty_cache()\n            del img\n            del mask\n    TOTAL_VALIDATION_LOSS.append(np.mean(temp_valid_loss))\n\n    # Print statistics for each epoch\n    print(f'Epoch [{epoch + 1}/{EPOCHS}], '\n          f'Train Loss: {TOTAL_TRAIN_LOSS[-1]:.4f}, '\n          f'Validation Loss: {TOTAL_VALIDATION_LOSS[-1]:.4f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T11:06:55.053986Z","iopub.status.idle":"2024-10-14T11:06:55.054314Z","shell.execute_reply.started":"2024-10-14T11:06:55.054142Z","shell.execute_reply":"2024-10-14T11:06:55.054158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## save the model \n# torch.save(unet.state_dict(), '/kaggle/working/model_state_dict.pth')\n# torch.save(optimizer.state_dict(), '/kaggle/working/optimizer_state_dict.pth')","metadata":{"execution":{"iopub.status.busy":"2024-10-14T11:06:55.055916Z","iopub.status.idle":"2024-10-14T11:06:55.056281Z","shell.execute_reply.started":"2024-10-14T11:06:55.056099Z","shell.execute_reply":"2024-10-14T11:06:55.056117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference Test images","metadata":{}},{"cell_type":"code","source":"# load the train model \n\n# unet.load_state_dict(torch.load('/kaggle/input/t1/pytorch/default/1/model_state_dict.pth'))\n# optimizer.load_state_dict(torch.load('/kaggle/input/t1/pytorch/default/1/optimizer_state_dict.pth'))","metadata":{"execution":{"iopub.status.busy":"2024-10-15T06:49:03.083637Z","iopub.execute_input":"2024-10-15T06:49:03.084338Z","iopub.status.idle":"2024-10-15T06:49:03.197732Z","shell.execute_reply.started":"2024-10-15T06:49:03.084300Z","shell.execute_reply":"2024-10-15T06:49:03.196714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## TEST_DIR is the array of the paths for test images\n\n\nwith zipfile.ZipFile(TEST_DIR) as z:\n    z.extractall()\n    \nTEST_DIR_BASE_PATH = \"/kaggle/working/test\"\nTEST_DIR = [os.path.join(TEST_DIR_BASE_PATH, i) for i in os.listdir(TEST_DIR_BASE_PATH)]\nprint(len(TEST_DIR))","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:23:42.426894Z","iopub.execute_input":"2024-10-15T07:23:42.427883Z","iopub.status.idle":"2024-10-15T07:23:42.778870Z","shell.execute_reply.started":"2024-10-15T07:23:42.427832Z","shell.execute_reply":"2024-10-15T07:23:42.777819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomTestDataset(Dataset):\n    def __init__(self, img_paths, transform):\n        self.img_paths = img_paths \n        self.transform = transform\n    \n    def __len__(self):\n        return len(self.img_paths)\n    \n    def __getitem__(self, idx):\n        img_path = self.img_paths[idx]\n        img_name = img_path.split('/')[-1]\n        img = cv2.imread(img_path)\n        if(transform):\n            img = self.transform(img)\n        return img_name, img","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:24:01.944881Z","iopub.execute_input":"2024-10-15T07:24:01.945290Z","iopub.status.idle":"2024-10-15T07:24:01.952655Z","shell.execute_reply.started":"2024-10-15T07:24:01.945244Z","shell.execute_reply":"2024-10-15T07:24:01.951488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### While inferencing on test dataset, it contains almost 100000 samples. so its impossible to load all the samples and work on them. \n\n#### the solution which helps me solve this issue\n    - use lower batch size\n    - use num_workers to asynchornously laod the data\n    - After certain iteration, store the temporary output and clear the memory    \n ","metadata":{}},{"cell_type":"code","source":"testDataset = CustomTestDataset(TEST_DIR, transform) \ntestLoader = DataLoader(testDataset, batch_size=4, shuffle=True, num_workers=4)\ntmp = next(iter(testLoader))","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:24:28.147286Z","iopub.execute_input":"2024-10-15T07:24:28.148359Z","iopub.status.idle":"2024-10-15T07:24:29.051925Z","shell.execute_reply.started":"2024-10-15T07:24:28.148301Z","shell.execute_reply":"2024-10-15T07:24:29.050702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(f\"Images on device: {imgs.device}, Model on device: {next(unet.parameters()).device}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:24:37.967615Z","iopub.execute_input":"2024-10-15T07:24:37.968419Z","iopub.status.idle":"2024-10-15T07:24:37.972631Z","shell.execute_reply.started":"2024-10-15T07:24:37.968380Z","shell.execute_reply":"2024-10-15T07:24:37.971470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(columns=['img', 'rle_mask'])\nunet.eval()\n\nTEST_MASK = []  \n\nfor batch_idx, (img_names, imgs) in enumerate(tqdm.tqdm(testLoader)):\n    imgs = imgs.to(DEVICE)\n    masks = unet(imgs)\n    for img_name, mask in zip(img_names, masks):\n        rle_encoded_mask = rle_encode(mask.cpu()) \n        TEST_MASK.append([img_name, rle_encoded_mask])\n    del imgs, masks\n    torch.cuda.empty_cache()\n#     print(torch.cuda.memory_summary(device=DEVICE))\n\n    # Save results every 5 batches\n    if batch_idx % 5 == 0:\n        tmp = pd.DataFrame(TEST_MASK, columns=['img', 'rle_mask'])\n        sub = pd.concat([sub, tmp], ignore_index=True)  \n        TEST_MASK = []  \n        \n\n# If there are remaining results in TEST_MASK after the loop\nif TEST_MASK:\n    tmp = pd.DataFrame(TEST_MASK, columns=['img', 'rle_mask'])\n    sub = pd.concat([sub, tmp], ignore_index=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:40:53.950482Z","iopub.execute_input":"2024-10-15T07:40:53.950895Z","iopub.status.idle":"2024-10-15T08:35:42.785771Z","shell.execute_reply.started":"2024-10-15T07:40:53.950856Z","shell.execute_reply":"2024-10-15T08:35:42.784512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(os.path.join(\"/kaggle/working/\", 'submission.csv'), index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:35:42.787569Z","iopub.execute_input":"2024-10-15T08:35:42.787904Z","iopub.status.idle":"2024-10-15T08:37:54.323866Z","shell.execute_reply.started":"2024-10-15T08:35:42.787866Z","shell.execute_reply":"2024-10-15T08:37:54.322941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom IPython.display import FileLink\n\nos.chdir('/kaggle/working')\nFileLink('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:37:54.331674Z","iopub.execute_input":"2024-10-15T08:37:54.332368Z","iopub.status.idle":"2024-10-15T08:37:54.343302Z","shell.execute_reply.started":"2024-10-15T08:37:54.332325Z","shell.execute_reply":"2024-10-15T08:37:54.342456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:48:53.717889Z","iopub.execute_input":"2024-10-15T08:48:53.718294Z","iopub.status.idle":"2024-10-15T08:48:53.725188Z","shell.execute_reply.started":"2024-10-15T08:48:53.718254Z","shell.execute_reply":"2024-10-15T08:48:53.724007Z"},"trusted":true},"execution_count":null,"outputs":[]}]}