{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":6927,"databundleVersionId":45059}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n\nimport os\n\nprint(os.listdir(\"/kaggle/input/competitions\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:53.209112Z","iopub.execute_input":"2026-05-20T19:48:53.209840Z","iopub.status.idle":"2026-05-20T19:48:53.214731Z","shell.execute_reply.started":"2026-05-20T19:48:53.209807Z","shell.execute_reply":"2026-05-20T19:48:53.213918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir(\"/kaggle/input/competitions/carvana-image-masking-challenge\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:53.216610Z","iopub.execute_input":"2026-05-20T19:48:53.216987Z","iopub.status.idle":"2026-05-20T19:48:53.233691Z","shell.execute_reply.started":"2026-05-20T19:48:53.216948Z","shell.execute_reply":"2026-05-20T19:48:53.232683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\n\nzip_path = \"/kaggle/input/competitions/carvana-image-masking-challenge/train.zip\"\n\nextract_path = \"/kaggle/working/train\"\n\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_path)\n\nprint(\"Train images extracted!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:53.234856Z","iopub.execute_input":"2026-05-20T19:48:53.235571Z","iopub.status.idle":"2026-05-20T19:48:58.376124Z","shell.execute_reply.started":"2026-05-20T19:48:53.235526Z","shell.execute_reply":"2026-05-20T19:48:58.375098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"zip_path = \"/kaggle/input/competitions/carvana-image-masking-challenge/train_masks.zip\"\n\nextract_path = \"/kaggle/working/train_masks\"\n\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_path)\n\nprint(\"Masks extracted!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:58.378367Z","iopub.execute_input":"2026-05-20T19:48:58.378689Z","iopub.status.idle":"2026-05-20T19:48:59.284189Z","shell.execute_reply.started":"2026-05-20T19:48:58.378663Z","shell.execute_reply":"2026-05-20T19:48:59.283344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(os.listdir(\"/kaggle/working/train\")))\nprint(len(os.listdir(\"/kaggle/working/train_masks\")))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:59.285157Z","iopub.execute_input":"2026-05-20T19:48:59.285513Z","iopub.status.idle":"2026-05-20T19:48:59.292904Z","shell.execute_reply.started":"2026-05-20T19:48:59.285487Z","shell.execute_reply":"2026-05-20T19:48:59.291993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport torch\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport torch.nn as nn\n\nfrom tqdm import tqdm\nfrom torch.utils.data import Dataset, DataLoader","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:59.294190Z","iopub.execute_input":"2026-05-20T19:48:59.294550Z","iopub.status.idle":"2026-05-20T19:48:59.304545Z","shell.execute_reply.started":"2026-05-20T19:48:59.294507Z","shell.execute_reply":"2026-05-20T19:48:59.303666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nprint(DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:59.305741Z","iopub.execute_input":"2026-05-20T19:48:59.306068Z","iopub.status.idle":"2026-05-20T19:48:59.321102Z","shell.execute_reply.started":"2026-05-20T19:48:59.306034Z","shell.execute_reply":"2026-05-20T19:48:59.320076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\n\n# =========================================================\n# DOUBLE CONVOLUTION BLOCK\n# =========================================================\nclass DoubleConv(nn.Module):\n\n    def __init__(self, in_channels, out_channels, dropout=0.0):\n        super().__init__()\n\n        self.double_conv = nn.Sequential(\n\n            # First Convolution\n            nn.Conv2d(\n                in_channels=in_channels,\n                out_channels=out_channels,\n                kernel_size=3,\n                padding=1,\n                bias=False\n            ),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n\n            # Optional Dropout\n            nn.Dropout2d(dropout),\n\n            # Second Convolution\n            nn.Conv2d(\n                in_channels=out_channels,\n                out_channels=out_channels,\n                kernel_size=3,\n                padding=1,\n                bias=False\n            ),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True)\n        )\n\n    def forward(self, x):\n        return self.double_conv(x)\n\n\n# =========================================================\n# ENCODER BLOCK\n# =========================================================\nclass EncoderBlock(nn.Module):\n\n    def __init__(self, in_channels, out_channels, dropout=0.0):\n        super().__init__()\n\n        self.conv = DoubleConv(\n            in_channels,\n            out_channels,\n            dropout\n        )\n\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2)\n\n    def forward(self, x):\n\n        skip = self.conv(x)      # Feature map for skip connection\n        pooled = self.pool(skip) # Downsampled feature map\n\n        return skip, pooled\n\n\n# =========================================================\n# BOTTLENECK BLOCK\n# =========================================================\nclass Bottleneck(nn.Module):\n\n    def __init__(self, in_channels, out_channels):\n        super().__init__()\n\n        self.bottleneck = DoubleConv(\n            in_channels,\n            out_channels\n        )\n\n    def forward(self, x):\n        return self.bottleneck(x)\n\n\n# =========================================================\n# DECODER BLOCK\n# =========================================================\nclass DecoderBlock(nn.Module):\n\n    def __init__(self, in_channels, out_channels, dropout=0.0):\n        super().__init__()\n\n        # Upsampling using Transposed Convolution\n        self.up = nn.ConvTranspose2d(\n            in_channels,\n            out_channels,\n            kernel_size=2,\n            stride=2\n        )\n\n        # Double Convolution after concatenation\n        self.conv = DoubleConv(\n            out_channels * 2,\n            out_channels,\n            dropout\n        )\n\n    def forward(self, x, skip_connection):\n\n        # Upsample\n        x = self.up(x)\n\n        # =================================================\n        # SKIP CONNECTION HANDLING\n        # =================================================\n        # Concatenate encoder feature maps with decoder maps\n        x = torch.cat([x, skip_connection], dim=1)\n\n        # Convolution Refinement\n        x = self.conv(x)\n\n        return x\n\n\n# =========================================================\n# FINAL OUTPUT LAYER\n# =========================================================\nclass FinalOutputLayer(nn.Module):\n\n    def __init__(self, in_channels, out_channels):\n        super().__init__()\n\n        self.final_layer = nn.Conv2d(\n            in_channels,\n            out_channels,\n            kernel_size=1\n        )\n\n    def forward(self, x):\n        return self.final_layer(x)\n\n\n# =========================================================\n# MAIN U-NET MODEL\n# =========================================================\nclass UNet(nn.Module):\n\n    def __init__(self, in_channels=3, out_channels=1):\n        super().__init__()\n\n        # =================================================\n        # ENCODER PATH\n        # =================================================\n        self.encoder1 = EncoderBlock(in_channels, 64)\n        self.encoder2 = EncoderBlock(64, 128)\n        self.encoder3 = EncoderBlock(128, 256)\n        self.encoder4 = EncoderBlock(256, 512, dropout=0.3)\n\n        # =================================================\n        # BOTTLENECK\n        # =================================================\n        self.bottleneck = Bottleneck(512, 1024)\n\n        # =================================================\n        # DECODER PATH\n        # =================================================\n        self.decoder4 = DecoderBlock(1024, 512, dropout=0.3)\n        self.decoder3 = DecoderBlock(512, 256)\n        self.decoder2 = DecoderBlock(256, 128)\n        self.decoder1 = DecoderBlock(128, 64)\n\n        # =================================================\n        # FINAL OUTPUT\n        # =================================================\n        self.output = FinalOutputLayer(64, out_channels)\n\n    def forward(self, x):\n\n        # =================================================\n        # ENCODER\n        # =================================================\n        skip1, p1 = self.encoder1(x)\n\n        skip2, p2 = self.encoder2(p1)\n\n        skip3, p3 = self.encoder3(p2)\n\n        skip4, p4 = self.encoder4(p3)\n\n        # =================================================\n        # BOTTLENECK\n        # =================================================\n        bottleneck = self.bottleneck(p4)\n\n        # =================================================\n        # DECODER\n        # =================================================\n        d4 = self.decoder4(bottleneck, skip4)\n\n        d3 = self.decoder3(d4, skip3)\n\n        d2 = self.decoder2(d3, skip2)\n\n        d1 = self.decoder1(d2, skip1)\n\n        # =================================================\n        # FINAL OUTPUT\n        # =================================================\n        output = self.output(d1)\n\n        return output\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:59.322463Z","iopub.execute_input":"2026-05-20T19:48:59.322856Z","iopub.status.idle":"2026-05-20T19:48:59.340836Z","shell.execute_reply.started":"2026-05-20T19:48:59.322820Z","shell.execute_reply":"2026-05-20T19:48:59.339979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CarvanaDataset(Dataset):\n\n    def __init__(self, image_dir, mask_dir):\n\n        self.image_dir = image_dir\n        self.mask_dir = mask_dir\n\n        self.images = os.listdir(image_dir)\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, index):\n\n        image_name = self.images[index]\n\n        image_path = os.path.join(\n            self.image_dir,\n            image_name\n        )\n\n        mask_name = image_name.replace(\".jpg\", \"_mask.gif\")\n\n        mask_path = os.path.join(\n            self.mask_dir,\n            mask_name\n        )\n\n        image = cv2.imread(image_path)\n\n        image = cv2.cvtColor(\n            image,\n            cv2.COLOR_BGR2RGB\n        )\n\n        mask = cv2.imread(\n            mask_path,\n            cv2.IMREAD_GRAYSCALE\n        )\n\n        image = cv2.resize(image, (256, 256))\n        mask = cv2.resize(mask, (256, 256))\n\n        image = image / 255.0\n        mask = mask / 255.0\n\n        image = np.transpose(image, (2, 0, 1))\n\n        image = torch.tensor(image, dtype=torch.float32)\n\n        mask = torch.tensor(mask, dtype=torch.float32)\n\n        mask = mask.unsqueeze(0)\n\n        return image, mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:48:59.341852Z","iopub.execute_input":"2026-05-20T19:48:59.342170Z","iopub.status.idle":"2026-05-20T19:48:59.356801Z","shell.execute_reply.started":"2026-05-20T19:48:59.342146Z","shell.execute_reply":"2026-05-20T19:48:59.355952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_DIR = \"/kaggle/working/train/train\"\n\nMASK_DIR = \"/kaggle/working/train_masks/train_masks\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:50:02.887767Z","iopub.execute_input":"2026-05-20T19:50:02.888440Z","iopub.status.idle":"2026-05-20T19:50:02.892417Z","shell.execute_reply.started":"2026-05-20T19:50:02.888407Z","shell.execute_reply":"2026-05-20T19:50:02.891508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = CarvanaDataset(\n    IMAGE_DIR,\n    MASK_DIR\n)\n\nloader = DataLoader(\n    dataset,\n    batch_size=4,\n    shuffle=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:50:08.064304Z","iopub.execute_input":"2026-05-20T19:50:08.064580Z","iopub.status.idle":"2026-05-20T19:50:08.072653Z","shell.execute_reply.started":"2026-05-20T19:50:08.064558Z","shell.execute_reply":"2026-05-20T19:50:08.071920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = UNet().to(DEVICE)\n\nloss_fn = nn.BCEWithLogitsLoss()\n\noptimizer = torch.optim.Adam(\n    model.parameters(),\n    lr=1e-4\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:50:13.879367Z","iopub.execute_input":"2026-05-20T19:50:13.879798Z","iopub.status.idle":"2026-05-20T19:50:14.157977Z","shell.execute_reply.started":"2026-05-20T19:50:13.879769Z","shell.execute_reply":"2026-05-20T19:50:14.157219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 3\n\nfor epoch in range(EPOCHS):\n\n    loop = tqdm(loader)\n\n    for images, masks in loop:\n\n        images = images.to(DEVICE)\n        masks = masks.to(DEVICE)\n\n        predictions = model(images)\n\n        loss = loss_fn(predictions, masks)\n\n        optimizer.zero_grad()\n\n        loss.backward()\n\n        optimizer.step()\n\n        loop.set_description(f\"Epoch {epoch+1}\")\n\n        loop.set_postfix(loss=loss.item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T19:50:16.549559Z","iopub.execute_input":"2026-05-20T19:50:16.549996Z","iopub.status.idle":"2026-05-20T20:20:15.388124Z","shell.execute_reply.started":"2026-05-20T19:50:16.549967Z","shell.execute_reply":"2026-05-20T20:20:15.387393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(model.state_dict(), \"unet.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T20:20:35.063663Z","iopub.execute_input":"2026-05-20T20:20:35.063988Z","iopub.status.idle":"2026-05-20T20:20:35.268765Z","shell.execute_reply.started":"2026-05-20T20:20:35.063962Z","shell.execute_reply":"2026-05-20T20:20:35.268104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()\n\nimage, mask = dataset[0]\n\nwith torch.no_grad():\n\n    prediction = model(\n        image.unsqueeze(0).to(DEVICE)\n    )\n\n    prediction = torch.sigmoid(prediction)\n\n    prediction = (prediction > 0.5).float()\n\n\npred_mask = prediction.squeeze().cpu().numpy()\n\nimage = image.permute(1,2,0).numpy()\n\nreal_mask = mask.squeeze().numpy()\n\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1,3,1)\nplt.imshow(image)\nplt.title(\"Image\")\n\nplt.subplot(1,3,2)\nplt.imshow(real_mask, cmap=\"gray\")\nplt.title(\"Ground Truth\")\n\nplt.subplot(1,3,3)\nplt.imshow(pred_mask, cmap=\"gray\")\nplt.title(\"Prediction\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-20T20:20:44.554580Z","iopub.execute_input":"2026-05-20T20:20:44.555415Z","iopub.status.idle":"2026-05-20T20:20:45.302127Z","shell.execute_reply.started":"2026-05-20T20:20:44.555381Z","shell.execute_reply":"2026-05-20T20:20:45.301235Z"}},"outputs":[],"execution_count":null}]}