{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":92860,"databundleVersionId":11074257,"sourceType":"competition"}],"dockerImageVersionId":30887,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# SLICE MY FACE","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I trained a **DeepLabV3+** model for facial feature segmentation.  \nTrained the model on **3500 train images** and generated **RLE-encoded masks** for test images.\n\n","metadata":{}},{"cell_type":"markdown","source":"**IMPORTING THE REQUIRED LIBRARIES**\n1. pandas, numpy, os for file and data management\n2. PIL.Image, cv2 for handling and augmenting images\n3. Torch, torchvision.transforms for model training and preprocessing\n4. tqmd for progress bar\n5. DataLoader loads data in batches with shuffling and multiprocessing.","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:09:52.674685Z","iopub.execute_input":"2025-02-25T14:09:52.674957Z","iopub.status.idle":"2025-02-25T14:09:58.902864Z","shell.execute_reply.started":"2025-02-25T14:09:52.674935Z","shell.execute_reply":"2025-02-25T14:09:58.902178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Defining the path of the datasets**\n\nDataset path is stored in DATASET_PATH.\nThe Dataset is divided into two parts- annotations and images, which are further divided into train,test and val.\nTRAIN_IMG_DIR contains the images and TRAIN_MASK_DIR contains the masks on which the model is trained, then the model predicts the mask values for the images stored in TEST_IMG_DIR.\n","metadata":{}},{"cell_type":"code","source":"DATASET_PATH = \"/kaggle/input/slicee-my-face\"\nTRAIN_IMG_DIR = os.path.join(DATASET_PATH, \"images/train\")\nTRAIN_MASK_DIR = os.path.join(DATASET_PATH, \"annotations/train\")\nTEST_IMG_DIR = os.path.join(DATASET_PATH, \"images/test\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:09:58.903901Z","iopub.execute_input":"2025-02-25T14:09:58.904282Z","iopub.status.idle":"2025-02-25T14:09:58.908051Z","shell.execute_reply.started":"2025-02-25T14:09:58.904225Z","shell.execute_reply":"2025-02-25T14:09:58.907263Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"IMG_SIZE = (256,256): Standardizes all images and masks to 256×256 pixels for uniformity.\n\nCUDA Support: Detects if a GPU is available and assigns the device.\n\nDefined the batch_size=8(all the 3500 are divided into 8 batches for training),epochs=5(number of iterations) and learning rate=0.0001.","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = (256, 256)  # Resize images\nBATCH_SIZE = 8\nEPOCHS = 5\nLR = 1e-4  # Learning rate\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:09:58.909493Z","iopub.execute_input":"2025-02-25T14:09:58.909805Z","iopub.status.idle":"2025-02-25T14:09:58.988848Z","shell.execute_reply.started":"2025-02-25T14:09:58.909745Z","shell.execute_reply":"2025-02-25T14:09:58.988106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Custom Dataset Class for Facial Segmentation\n\nThis class loads the images and masks for training.  \nFor test images, only images are loaded.\n\n\n\n","metadata":{}},{"cell_type":"code","source":"class FaceSegmentationDataset(Dataset):\n    def __init__(self, img_dir, mask_dir=None, transform=None, train=True):\n        self.img_dir = img_dir\n        self.mask_dir = mask_dir\n        self.image_files = sorted(os.listdir(img_dir))\n        self.transform = transform\n        self.train = train\n\n    def __len__(self):\n        return len(self.image_files)\n\n    def __getitem__(self, idx):\n        img_name = self.image_files[idx]\n        img_path = os.path.join(self.img_dir, img_name)\n        \n        # Load and preprocess image\n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        if self.train:\n            # Load mask if in training mode\n            mask_path = os.path.join(self.mask_dir, img_name.replace(\".jpg\", \".png\"))\n            mask = cv2.imread(mask_path, cv2.IMREAD_GRAYSCALE)\n            mask = (mask > 0).astype(np.uint8)  # Ensure binary mask\n\n            # Apply transformations\n            if self.transform:\n                transformed = self.transform(image=image, mask=mask)\n                image, mask = transformed[\"image\"], transformed[\"mask\"]\n            \n            return image, mask\n        \n        else:\n            # For test images, return only image\n            if self.transform:\n                image = self.transform(image=image)[\"image\"]\n            return image, img_name","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:09:58.990011Z","iopub.execute_input":"2025-02-25T14:09:58.990228Z","iopub.status.idle":"2025-02-25T14:09:59.128688Z","shell.execute_reply.started":"2025-02-25T14:09:58.990210Z","shell.execute_reply":"2025-02-25T14:09:59.127818Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Transformations using Albumentations\nResized the images, normalized them, and applied basic augmentations.\n","metadata":{}},{"cell_type":"code","source":"import albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\ntrain_transform = A.Compose([\n    A.Resize(IMG_SIZE[0], IMG_SIZE[1]),\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(p=0.2),\n    A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n    ToTensorV2(),\n])\n\ntest_transform = A.Compose([\n    A.Resize(IMG_SIZE[0], IMG_SIZE[1]),\n    A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n    ToTensorV2(),\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:09:59.129555Z","iopub.execute_input":"2025-02-25T14:09:59.129853Z","iopub.status.idle":"2025-02-25T14:10:00.490316Z","shell.execute_reply.started":"2025-02-25T14:09:59.129831Z","shell.execute_reply":"2025-02-25T14:10:00.489365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Loading the Training and Test Datasets","metadata":{}},{"cell_type":"code","source":"train_dataset = FaceSegmentationDataset(TRAIN_IMG_DIR, TRAIN_MASK_DIR, transform=train_transform, train=True)\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\n\ntest_dataset = FaceSegmentationDataset(TEST_IMG_DIR, transform=test_transform, train=False)\ntest_loader = DataLoader(test_dataset, batch_size=1, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:10:00.491327Z","iopub.execute_input":"2025-02-25T14:10:00.491775Z","iopub.status.idle":"2025-02-25T14:10:00.859741Z","shell.execute_reply.started":"2025-02-25T14:10:00.491737Z","shell.execute_reply":"2025-02-25T14:10:00.858742Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DeepLabV3+ Model\nUsed DeepLabV3+ with a modified classifier for binary segmentation and **Dice Loss + BCE Loss** for training.\n","metadata":{}},{"cell_type":"code","source":"class DeepLabV3Plus(nn.Module):\n    def __init__(self, num_classes=1):\n        super(DeepLabV3Plus, self).__init__()\n        self.model = models.segmentation.deeplabv3_resnet50(pretrained=True)\n        self.model.classifier[4] = nn.Conv2d(256, num_classes, kernel_size=1)  # Modify output layer\n\n    def forward(self, x):\n        return self.model(x)[\"out\"]\n\n# Initialize model\nmodel = DeepLabV3Plus().to(DEVICE)\n\n# Loss & Optimizer\ndef dice_loss(pred, target, smooth=1e-6):\n    pred = torch.sigmoid(pred)\n    intersection = (pred * target).sum()\n    return 1 - (2. * intersection + smooth) / (pred.sum() + target.sum() + smooth)\n\ncriterion = lambda pred, target: 0.5 * dice_loss(pred, target) + 0.5 * nn.BCEWithLogitsLoss()(pred, target)\noptimizer = optim.AdamW(model.parameters(), lr=LR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:10:00.860679Z","iopub.execute_input":"2025-02-25T14:10:00.860991Z","iopub.status.idle":"2025-02-25T14:10:02.851203Z","shell.execute_reply.started":"2025-02-25T14:10:00.860961Z","shell.execute_reply":"2025-02-25T14:10:02.850515Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training Loop\nTrained for 5 epochs and printed the loss after each epoch.","metadata":{}},{"cell_type":"code","source":"for epoch in range(EPOCHS):\n    model.train()\n    epoch_loss = 0\n    for images, masks in tqdm(train_loader, desc=f\"Epoch {epoch+1}/{EPOCHS}\"):\n        images, masks = images.to(DEVICE), masks.to(DEVICE).float().unsqueeze(1)\n\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, masks)\n        loss.backward()\n        optimizer.step()\n\n        epoch_loss += loss.item()\n    \n    print(f\"Epoch {epoch+1} Loss: {epoch_loss / len(train_loader):.4f}\")\n\n# Save Model\ntorch.save(model.state_dict(), \"deeplabv3_model.pth\")\nprint(\"Model saved! ✅\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:10:02.853444Z","iopub.execute_input":"2025-02-25T14:10:02.853692Z","iopub.status.idle":"2025-02-25T14:36:37.048779Z","shell.execute_reply.started":"2025-02-25T14:10:02.853671Z","shell.execute_reply":"2025-02-25T14:36:37.047844Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Code to convert the mask into rle.","metadata":{}},{"cell_type":"code","source":"def rle_encode(mask):\n    \"\"\"Convert binary mask to RLE format.\"\"\"\n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])  # Add padding\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return \" \".join(str(x) for x in runs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:36:37.049907Z","iopub.execute_input":"2025-02-25T14:36:37.050205Z","iopub.status.idle":"2025-02-25T14:36:37.054937Z","shell.execute_reply.started":"2025-02-25T14:36:37.050182Z","shell.execute_reply":"2025-02-25T14:36:37.054090Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Generate Predictions and Save Submission File\nPredicted masks for test images and encode them using RLE.","metadata":{}},{"cell_type":"code","source":"model.eval()\nsubmission = []\n\nwith torch.no_grad():\n    for image, img_name in tqdm(test_loader, desc=\"Generating Predictions\"):\n        image = image.to(DEVICE)\n\n        # Predict mask\n        output = model(image)\n        pred_mask = torch.sigmoid(output).cpu().numpy().squeeze()\n\n        # Convert to binary mask (threshold = 0.5)\n        binary_mask = (pred_mask > 0.5).astype(np.uint8)\n\n        # Resize back to original dimensions\n        original_size = cv2.imread(os.path.join(TEST_IMG_DIR, img_name[0])).shape[:2]\n        binary_mask = cv2.resize(binary_mask, (original_size[1], original_size[0]), interpolation=cv2.INTER_NEAREST)\n\n        # Encode as RLE\n        rle_mask = rle_encode(binary_mask)\n        submission.append([img_name[0].split(\".\")[0], rle_mask])\n\n# Save to CSV\nsubmission_df = pd.DataFrame(submission, columns=[\"id\", \"predicted\"])\nsubmission_df.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file saved as submission.csv ✅\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-25T14:36:37.055776Z","iopub.execute_input":"2025-02-25T14:36:37.056074Z","iopub.status.idle":"2025-02-25T14:38:17.725939Z","shell.execute_reply.started":"2025-02-25T14:36:37.056034Z","shell.execute_reply":"2025-02-25T14:38:17.725204Z"}},"outputs":[],"execution_count":null}]}