{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8700746,"sourceType":"datasetVersion","datasetId":5200442},{"sourceId":65938,"sourceType":"modelInstanceVersion","modelInstanceId":54421}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport cv2\nimport numpy as np\nimport os\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport albumentations as A\nimport torchvision\nfrom albumentations.pytorch import ToTensorV2\nfrom tqdm import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport torchvision.transforms.functional as TF\nimport zipfile\nfrom torchvision.io import read_image\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:00.806332Z","iopub.execute_input":"2024-06-17T06:01:00.806824Z","iopub.status.idle":"2024-06-17T06:01:00.814158Z","shell.execute_reply.started":"2024-06-17T06:01:00.806792Z","shell.execute_reply":"2024-06-17T06:01:00.812984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/*","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:00.816485Z","iopub.execute_input":"2024-06-17T06:01:00.816972Z","iopub.status.idle":"2024-06-17T06:01:01.954304Z","shell.execute_reply.started":"2024-06-17T06:01:00.816936Z","shell.execute_reply":"2024-06-17T06:01:01.952865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the U-Net architecture\nclass DoubleConv(nn.Module):\n    def __init__(self, in_channels, out_channels):\n        super(DoubleConv, self).__init__()\n        self.conv = nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, 3, 1, 1, bias=False),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_channels, out_channels, 3, 1, 1, bias=False),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n        )\n\n    def forward(self, x):\n        return self.conv(x)\n\nclass UNET(nn.Module):\n    def __init__(\n            self, in_channels=3, out_channels=1, features=[64, 128, 256, 512],\n    ):\n        super(UNET, self).__init__()\n        self.ups = nn.ModuleList()\n        self.downs = nn.ModuleList()\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2)\n\n        # Down part of UNET\n        for feature in features:\n            self.downs.append(DoubleConv(in_channels, feature))\n            in_channels = feature\n\n        # Up part of UNET\n        for feature in reversed(features):\n            self.ups.append(\n                nn.ConvTranspose2d(\n                    feature*2, feature, kernel_size=2, stride=2,\n                )\n            )\n            self.ups.append(DoubleConv(feature*2, feature))\n\n        self.bottleneck = DoubleConv(features[-1], features[-1]*2)\n        self.final_conv = nn.Conv2d(features[0], out_channels, kernel_size=1)\n\n    def forward(self, x):\n        skip_connections = []\n\n        for down in self.downs:\n            x = down(x)\n            skip_connections.append(x)\n            x = self.pool(x)\n\n        x = self.bottleneck(x)\n        skip_connections = skip_connections[::-1]\n\n        for idx in range(0, len(self.ups), 2):\n            x = self.ups[idx](x)\n            skip_connection = skip_connections[idx//2]\n\n            if x.shape != skip_connection.shape:\n                x = TF.resize(x, size=skip_connection.shape[2:])\n\n            concat_skip = torch.cat((skip_connection, x), dim=1)\n            x = self.ups[idx+1](concat_skip)\n\n        return self.final_conv(x)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:01.956448Z","iopub.execute_input":"2024-06-17T06:01:01.956935Z","iopub.status.idle":"2024-06-17T06:01:01.974958Z","shell.execute_reply.started":"2024-06-17T06:01:01.956888Z","shell.execute_reply":"2024-06-17T06:01:01.973990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparameters etc.\nLEARNING_RATE = 1e-4\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nBATCH_SIZE = 16\nNUM_EPOCHS = 10\nNUM_WORKERS = 2\nIMAGE_HEIGHT = 416  # 1280 originally\nIMAGE_WIDTH = 416  # 1918 originally \nPIN_MEMORY = True\nLOAD_MODEL = os.path.isfile(\"/kaggle/input/windowdetection/pytorch/window-detection/2/my_checkpoint.pth.tar\")\nWORKING_DIR = '/kaggle/working/'\n# EVAL_DIR = '/kaggle/input/windshield-detection/eval/eval'\nTRAIN_IMG_DIR = f\"/kaggle/input/windshield-detection/City Car/\"\nTRAIN_ANNOTATION = f\"{TRAIN_IMG_DIR}/annotation.csv\"\nTRAIN_MASK_DIR = f\"{WORKING_DIR}train_masks/\"\nVAL_IMG_DIR = f\"{WORKING_DIR}valid/\"\nVAL_MASK_DIR = f\"{WORKING_DIR}valid_maks/\"\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:45:15.926647Z","iopub.execute_input":"2024-06-17T06:45:15.927513Z","iopub.status.idle":"2024-06-17T06:45:15.938098Z","shell.execute_reply.started":"2024-06-17T06:45:15.927478Z","shell.execute_reply":"2024-06-17T06:45:15.937150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define the dataset class\nclass SegmentationDataset(Dataset):\n    def __init__(self, image_dir, mask_dir=None, transform=None):\n        self.image_dir = image_dir\n        self.mask_dir = mask_dir\n        self.transform = transform\n\n        # Get the list of image filenames\n        self.image_filenames = [f for f in os.listdir(image_dir) if f.endswith('.jpg')]  # Adjust extension if needed\n\n        # Filter out images without corresponding masks if mask_dir is provided\n        if self.mask_dir is not None:\n            self.image_filenames = [f for f in self.image_filenames \n                                    if os.path.exists(os.path.join(self.mask_dir, f.replace('.jpg', '_mask.png')))]  # Adjust file extension if needed\n\n    def __len__(self):\n        return len(self.image_filenames)\n\n    def __getitem__(self, idx):\n        image_filename = self.image_filenames[idx]\n        img_path = os.path.join(self.image_dir, image_filename)\n\n        # Load image\n        image = Image.open(img_path).convert('RGB')\n\n        # Load mask\n        mask_path = os.path.join(self.mask_dir, image_filename.replace('.jpg', '_mask.png'))  # Adjust file extension if needed\n        mask = Image.open(mask_path).convert('L')\n        image = np.array(Image.open(img_path).convert(\"RGB\"))\n        mask = np.array(Image.open(mask_path).convert(\"L\"), dtype=np.float32)\n        mask[mask == 255.0] = 1.0\n\n        # Apply transformations\n        if self.transform is not None:\n            augmentations = self.transform(image=image, mask=mask)\n            image = augmentations[\"image\"]\n            mask = augmentations[\"mask\"]\n            \n        return image, mask\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:02.006245Z","iopub.execute_input":"2024-06-17T06:01:02.006587Z","iopub.status.idle":"2024-06-17T06:01:02.018880Z","shell.execute_reply.started":"2024-06-17T06:01:02.006553Z","shell.execute_reply":"2024-06-17T06:01:02.017848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_masks(data_dir, annotations_file, mask_dir):\n    # Load annotations\n    annotations = pd.read_csv(annotations_file)\n\n    # Create mask folder if it doesn't exist\n    if not os.path.exists(mask_dir):\n        os.makedirs(mask_dir)\n\n    # Generate masks for each image\n    for _, row in annotations.iterrows():\n        image_path = os.path.join(data_dir, row['image_name'])\n        image = cv2.imread(image_path)\n\n        # Extract bounding box coordinates and dimensions\n        bbox_x = int(row['bbox_x'])\n        bbox_y = int(row['bbox_y'])\n        bbox_width = int(row['bbox_width'])\n        bbox_height = int(row['bbox_height'])\n\n        # Calculate xmax and ymax based on coordinates and dimensions\n        xmax = bbox_x + bbox_width\n        ymax = bbox_y + bbox_height\n\n        # Create a blank mask\n        mask = np.zeros(image.shape[:2], dtype=np.uint8)\n\n        # Fill the mask with white color within the bounding box\n        cv2.rectangle(mask, (bbox_x, bbox_y), (xmax, ymax), 255, -1)\n\n        # Save the mask as a grayscale image\n        mask_path = os.path.join(mask_dir, row['image_name'].replace('.jpg', '_mask.png'))\n        cv2.imwrite(mask_path, mask)\n        \n# Define data paths\ndata_dirs = [\n    TRAIN_IMG_DIR\n]\nannotations_files = [\n    TRAIN_ANNOTATION\n]\nmask_dirs = [\n    TRAIN_MASK_DIR\n]\n\n# Iterate through datasets and generate masks\nfor data_dir, annotations_file, mask_dir in zip(data_dirs, annotations_files, mask_dirs):\n    generate_masks(data_dir, annotations_file, mask_dir)\n\nprint(\"Mask generation completed!\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:02.020299Z","iopub.execute_input":"2024-06-17T06:01:02.020671Z","iopub.status.idle":"2024-06-17T06:01:04.215418Z","shell.execute_reply.started":"2024-06-17T06:01:02.020637Z","shell.execute_reply":"2024-06-17T06:01:04.214417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_LIST = [f for f in os.listdir(TRAIN_IMG_DIR) if f.endswith('.jpg')] \nTRAIN_LIST = [f for f in TRAIN_LIST if os.path.exists(os.path.join(TRAIN_MASK_DIR, f.replace('.jpg', '_mask.png')))]\n\nprint(\n        len(TRAIN_LIST),\n        len(os.listdir(TRAIN_MASK_DIR))\n    )\n\nos.mkdir(VAL_IMG_DIR)\nfor file in sorted(TRAIN_LIST)[580:]: \n    shutil.copy(TRAIN_IMG_DIR + file, VAL_IMG_DIR)\n\nos.mkdir(VAL_MASK_DIR)\nfor file in sorted(os.listdir(TRAIN_MASK_DIR))[580:]:\n    shutil.move(TRAIN_MASK_DIR + file, VAL_MASK_DIR)\n    \nos.mkdir(WORKING_DIR + 'saved_images')","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:04.216741Z","iopub.execute_input":"2024-06-17T06:01:04.217123Z","iopub.status.idle":"2024-06-17T06:01:04.312833Z","shell.execute_reply.started":"2024-06-17T06:01:04.217087Z","shell.execute_reply":"2024-06-17T06:01:04.311839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_checkpoint(state, filename=f\"{WORKING_DIR}/my_checkpoint.pth.tar\"):\n    print(\"=> Saving checkpoint\")\n    torch.save(state, filename)\n\ndef load_checkpoint(checkpoint, model):\n    print(\"=> Loading checkpoint\")\n    model.load_state_dict(checkpoint[\"state_dict\"])\n\ndef get_loaders(\n    train_dir, train_maskdir, val_dir, val_maskdir, batch_size, train_transform, val_transform, num_workers=4, pin_memory=True,):\n    \n    train_ds = SegmentationDataset(\n        image_dir=train_dir,\n        mask_dir=train_maskdir,\n        transform=train_transform,\n    )\n\n    train_loader = DataLoader(\n        train_ds,\n        batch_size=batch_size,\n        num_workers=num_workers,\n        pin_memory=pin_memory,\n        shuffle=True,\n    )\n\n    val_ds = SegmentationDataset(\n        image_dir=val_dir,\n        mask_dir=val_maskdir,\n        transform=val_transform,\n    )\n\n    val_loader = DataLoader(\n        val_ds,\n        batch_size=batch_size,\n        num_workers=num_workers,\n        pin_memory=pin_memory,\n        shuffle=False,\n    )\n\n    return train_loader, val_loader\n\ndef check_accuracy(loader, model, device=\"cuda\"):\n    num_correct = 0\n    num_pixels = 0\n    dice_score = 0\n    model.eval()\n\n    with torch.no_grad():\n        for x, y in loader:\n            x = x.to(device)\n            y = y.to(device).unsqueeze(1)\n            preds = torch.sigmoid(model(x))\n            preds = (preds > 0.5).float()\n            num_correct += (preds == y).sum()\n            num_pixels += torch.numel(preds)\n            dice_score += (2 * (preds * y).sum()) / (\n                (preds + y).sum() + 1e-8\n            )\n\n    print(\n        f\"Got {num_correct}/{num_pixels} with acc {num_correct/num_pixels*100:.2f}\"\n    )\n    print(f\"Dice score: {dice_score/len(loader)}\")\n    model.train()\n\ndef save_predictions_as_imgs(\n    loader, model, folder=\"kaggle/working/saved_images/\", device=\"cuda\"\n):\n    model.eval()\n    for idx, (x, y) in enumerate(loader):\n        x = x.to(device=device)\n        with torch.no_grad():\n            preds = torch.sigmoid(model(x))\n            preds = (preds > 0.5).float()\n        torchvision.utils.save_image(\n            preds, f\"{folder}/pred_{idx}.png\"\n        )\n        torchvision.utils.save_image(y.unsqueeze(1), f\"{folder}{idx}.png\")\n\n    model.train()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:04.314380Z","iopub.execute_input":"2024-06-17T06:01:04.314860Z","iopub.status.idle":"2024-06-17T06:01:04.331291Z","shell.execute_reply.started":"2024-06-17T06:01:04.314827Z","shell.execute_reply":"2024-06-17T06:01:04.330422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_fn(loader, model, optimizer, loss_fn, scaler):\n    loop = tqdm(loader)\n\n    for batch_idx, (data, targets) in enumerate(loop):\n        data = data.to(device=DEVICE)\n        targets = targets.float().unsqueeze(1).to(device=DEVICE)\n\n        # forward\n        with torch.cuda.amp.autocast():\n            predictions = model(data)\n            loss = loss_fn(predictions, targets)\n\n        # backward\n        optimizer.zero_grad()\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n\n        # update tqdm loop\n        loop.set_postfix(loss=loss.item())\n\n\ndef main():\n    train_transform = A.Compose(\n        [\n            A.Resize(height=IMAGE_HEIGHT, width=IMAGE_WIDTH),\n            A.Rotate(limit=35, p=1.0),\n            A.HorizontalFlip(p=0.5),\n            A.VerticalFlip(p=0.1),\n            A.Normalize(\n                mean=[0.0, 0.0, 0.0],\n                std=[1.0, 1.0, 1.0],\n                max_pixel_value=255.0,\n            ),\n            ToTensorV2(),\n        ],\n    )\n\n    val_transforms = A.Compose(\n        [\n            A.Resize(height=IMAGE_HEIGHT, width=IMAGE_WIDTH),\n            A.Normalize(\n                mean=[0.0, 0.0, 0.0],\n                std=[1.0, 1.0, 1.0],\n                max_pixel_value=255.0,\n            ),\n            ToTensorV2(),\n        ],\n    )\n\n    model = UNET(in_channels=3, out_channels=1).to(DEVICE)\n    loss_fn = nn.BCEWithLogitsLoss()\n    optimizer = optim.Adam(model.parameters(), lr=LEARNING_RATE)\n\n    train_loader, val_loader = get_loaders(\n        TRAIN_IMG_DIR,\n        TRAIN_MASK_DIR,\n        VAL_IMG_DIR,\n        VAL_MASK_DIR,\n        BATCH_SIZE,\n        train_transform,\n        val_transforms,\n        NUM_WORKERS,\n        PIN_MEMORY\n    )\n   \n    if LOAD_MODEL:\n        print(\"pre-trained\")\n        load_checkpoint(torch.load(f\"/kaggle/input/windowdetection/pytorch/window-detection/2/my_checkpoint.pth.tar\"), model)\n\n\n    check_accuracy(val_loader, model, device=DEVICE)\n    scaler = torch.cuda.amp.GradScaler()\n\n    for epoch in range(NUM_EPOCHS):\n        train_fn(train_loader, model, optimizer, loss_fn, scaler)\n\n        # save model\n        checkpoint = {\n            \"state_dict\": model.state_dict(),\n            \"optimizer\":optimizer.state_dict(),\n        }\n        save_checkpoint(checkpoint)\n\n        # check accuracy\n        check_accuracy(val_loader, model, device=DEVICE)\n        \n        save_predictions_as_imgs(\n            val_loader, model, folder=f\"{WORKING_DIR}\", device=DEVICE\n        )\n\nif __name__ == \"__main__\":\n    main()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:01:04.332897Z","iopub.execute_input":"2024-06-17T06:01:04.333292Z","iopub.status.idle":"2024-06-17T06:08:27.055572Z","shell.execute_reply.started":"2024-06-17T06:01:04.333257Z","shell.execute_reply":"2024-06-17T06:08:27.054036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport time\nimport urllib\n\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] -= runs[:-1:2]\n    \n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T06:08:27.058042Z","iopub.execute_input":"2024-06-17T06:08:27.059128Z","iopub.status.idle":"2024-06-17T06:08:27.066849Z","shell.execute_reply.started":"2024-06-17T06:08:27.059076Z","shell.execute_reply":"2024-06-17T06:08:27.065808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_DIR = WORKING_DIR + 'test'\n#os.mkdir(TEST_DIR)\nimage_path = f'{TEST_DIR}/temp_image.jpg'\nimage_file = \"https://ucarecdn.com/d051fb54-9074-49a0-a148-2576a2385ca2/IMG_1733.png\"\nurllib.request.urlretrieve(image_file, image_path)\n\nTHRESHOLD = 0.1\n\n# Dataset\nclass CarvanaTestDataset(Dataset):\n    def __init__(self, image_dir, transform=None):\n        self.image_dir = image_dir\n        self.transform = transform\n        self.images = sorted(os.listdir(image_dir))\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, index):\n        img_name = self.images[index]\n        img_path = os.path.join(self.image_dir, self.images[index])\n        image = np.array(Image.open(img_path).convert('RGB'))\n\n        if self.transform is not None:\n            augmentations = self.transform(image=image)\n            image = augmentations['image']\n\n        return img_name, image\n\n\ntest_transform = A.Compose(\n    [\n        A.Resize(height=IMAGE_HEIGHT, width=IMAGE_WIDTH),\n        A.Normalize(\n            mean=[0.0, 0.0, 0.0],\n            std=[1.0, 1.0, 1.0],\n            max_pixel_value=255.0,\n        ),\n        ToTensorV2(), \n    ]\n)\n   \ntest_set = CarvanaTestDataset(\n    image_dir= TRAIN_IMG_DIR,\n    transform=test_transform\n)    \n\n\ntest_loader = DataLoader(\n    test_set, batch_size=BATCH_SIZE, shuffle=False\n)\n    \n# Model\ncheckpoint = torch.load(WORKING_DIR + 'my_checkpoint.pth.tar')\n\nmodel = UNET(in_channels=3, out_channels=1).to(DEVICE)\nmodel.load_state_dict(checkpoint['state_dict'])\n\nmodel.eval()\n# Predictions\nall_predictions = []\nfor img_names, x in tqdm(test_loader):\n    x = x.to(DEVICE)\n    with torch.no_grad():\n        preds = torch.sigmoid(model(x))\n        preds = (preds > THRESHOLD).float()\n        size = x.shape[-2:]  # Get height and width from the original image\n\n    preds = TF.resize(\n        preds, size=size, interpolation=TF.InterpolationMode.NEAREST\n    )\n\n    # Encoding\n    for idx in range(len(img_names)):\n        encoding = rle_encode(preds[idx].squeeze().cpu())\n        all_predictions.append([img_names[idx], encoding])\n\n        # Visualization\n        pred_image = preds[idx].squeeze().cpu().numpy() * 255  # Scale to 0-255 for visualization\n\n        # Access the original image (after prediction and resize)\n        # Access the original image from the batch using the 'idx'\n        original_image = x[idx].permute(1, 2, 0).cpu().numpy()  # Move to CPU and rearrange dimensions\n\n        # Find contours in the mask (OpenCV)\n        contours, hierarchy = cv2.findContours(\n            pred_image.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE\n        )\n\n        # If contours are found, draw a bounding box\n        if len(contours) > 0:\n            # Find the largest contour (assuming it's the car)\n            largest_contour = max(contours, key=cv2.contourArea)\n            rect = cv2.minAreaRect(largest_contour)\n            box = cv2.boxPoints(rect).astype('int')\n            a, y, w, h = cv2.boundingRect(largest_contour)\n\n            # Draw the bounding box on the original image\n            original_image = (original_image * 255).astype(np.uint8)  # Convert to 0-255 range\n            original_image = cv2.cvtColor(original_image, cv2.COLOR_RGB2BGR)  # Convert to BGR for OpenCV\n            original_image = cv2.drawContours(original_image, [largest_contour], -1, (0, 255, 0), 2)  # Draw green bounding box\n\n        # Create a figure and axes for side-by-side display\n        fig, axes = plt.subplots(1, 2, figsize=(12, 6))\n        axes[0].imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))  # Display the image with the bounding box (convert back to RGB for Matplotlib)\n        axes[0].set_title(f\"Original Image: {img_names[idx]}\")\n        axes[1].imshow(pred_image, cmap='gray')\n        axes[1].set_title(f\"Predicted Mask: {img_names[idx]}\")\n        plt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T07:03:50.751033Z","iopub.execute_input":"2024-06-17T07:03:50.751459Z","iopub.status.idle":"2024-06-17T07:04:08.572271Z","shell.execute_reply.started":"2024-06-17T07:03:50.751429Z","shell.execute_reply":"2024-06-17T07:04:08.570375Z"},"trusted":true},"execution_count":null,"outputs":[]}]}