{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":85240,"databundleVersionId":9622164,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport torch\nimport pandas as pd\nimport cv2\nimport numpy as np\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import average_precision_score\nfrom tqdm import tqdm\nfrom PIL import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:03:55.806008Z","iopub.execute_input":"2024-12-05T12:03:55.806448Z","iopub.status.idle":"2024-12-05T12:04:02.596608Z","shell.execute_reply.started":"2024-12-05T12:03:55.806416Z","shell.execute_reply":"2024-12-05T12:04:02.594425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MosquitoDataset(Dataset):\n    def __init__(self, image_dir, label_dir, transform=None):\n        self.image_dir = image_dir\n        self.label_dir = label_dir\n        self.transform = transform\n        self.image_ids = [f.split('.')[0] for f in os.listdir(image_dir)]  # Image filenames without extensions\n    \n    def __len__(self):\n        return len(self.image_ids)\n    \n    def __getitem__(self, idx):\n        # Get image path and corresponding labels\n        image_id = self.image_ids[idx]\n        image_path = os.path.join(self.image_dir, f\"{image_id}.jpeg\")\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        # Load corresponding label\n        label_path = os.path.join(self.label_dir, f\"{image_id}.txt\")\n        with open(label_path, 'r') as file:\n            labels = file.readlines()\n        \n        # Parse labels into boxes and class labels\n        boxes = []\n        class_labels = []\n        for label in labels:\n            parts = label.strip().split()\n            class_label = int(parts[0])\n            xcenter, ycenter, width, height = map(float, parts[1:])\n            xmin = (xcenter - width / 2)\n            ymin = (ycenter - height / 2)\n            xmax = (xcenter + width / 2)\n            ymax = (ycenter + height / 2)\n            \n            # Convert normalized coordinates to absolute pixel values\n            boxes.append([xmin, ymin, xmax, ymax])\n            #boxes.append([xcenter, ycenter, width, height])\n            class_labels.append(class_label)\n        \n        # Convert to tensors\n        boxes = torch.tensor(boxes, dtype=torch.float32)\n        \n        class_labels = torch.tensor(class_labels, dtype=torch.int64)\n\n        target = {'image_id':image_id, 'boxes': boxes, 'labels': class_labels}  # Create the target dictionary\n\n        # Display the original image with bounding boxes (for debugging)\n        #print(boxes)\n        #self._show_image_with_boxes(image, boxes)\n        return image, target  # Ensure the return value is a tuple (image, target)\n\n    def _show_image_with_boxes(self, image, boxes):\n        \"\"\"Helper function to display the image with bounding boxes.\"\"\"\n        print(image.shape)\n        plt.figure(figsize=(8, 8))\n        plt.imshow(image)\n        ax = plt.gca()\n\n        for box in boxes:\n            xcenter, ycenter, width, height = box\n            # Convert to top-left corner coordinates\n            xmin = (xcenter - width / 2) * image.shape[1]\n            ymin = (ycenter - height / 2) * image.shape[0]\n            xmax = (xcenter + width / 2) * image.shape[1]\n            ymax = (ycenter + height / 2) * image.shape[0]\n            \n            # Draw bounding box\n            ax.add_patch(plt.Rectangle(\n                (xmin, ymin), xmax - xmin, ymax - ymin,\n                fill=False, color='red', linewidth=2\n            ))\n            #print(xmax)\n\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:04:02.598944Z","iopub.execute_input":"2024-12-05T12:04:02.599662Z","iopub.status.idle":"2024-12-05T12:04:02.618555Z","shell.execute_reply.started":"2024-12-05T12:04:02.599619Z","shell.execute_reply":"2024-12-05T12:04:02.616799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\n\ndef collate_fn(batch):\n    \"\"\" Custom collate function to pad images to the maximum size in the batch. \"\"\"\n    images, targets = zip(*batch)\n    \n    # Get the max width and height in the batch using image.shape instead of size\n    max_width = max([image.shape[1] for image in images])  # Width is at index 2\n    max_height = max([image.shape[0] for image in images])  # Height is at index 1\n\n    #print(images[0].shape, max_width, max_height)\n    padded_images = []\n    padded_targets = []\n\n    for image, target in zip(images, targets):\n        # Convert the image to a tensor if it is a numpy array\n        if isinstance(image, np.ndarray):\n            image = torch.from_numpy(image).permute(2, 0, 1)  # Convert and reorder dimensions (HWC -> CHW)\n\n        # Pad image\n        padding_width = max_width - image.shape[2]\n        padding_height = max_height - image.shape[1]\n\n        # Apply padding to the image (pad width and height)\n        #print(\"\\tbefore_padding: \", image.shape)\n        padded_image = torch.nn.functional.pad(image, (0, padding_width, 0, padding_height))\n        padded_image = padded_image.float()\n        padded_image = padded_image / 255.0\n        #print(\"\\tafter_padding\", padded_image.shape)\n        # Scale bounding boxes to match the new image size\n        boxes = target['boxes']\n        scale_factor_x = max_width / image.shape[2]\n        scale_factor_y = max_height / image.shape[1]\n        \n        # Scale the bounding boxes\n        scaled_boxes = boxes * torch.tensor([scale_factor_x, scale_factor_y, scale_factor_x, scale_factor_y])\n\n        # Update the target with the scaled bounding boxes\n        target['boxes'] = scaled_boxes\n\n        padded_images.append(padded_image)\n        padded_targets.append(target)\n\n    return torch.stack(padded_images), padded_targets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:04:02.625512Z","iopub.execute_input":"2024-12-05T12:04:02.626021Z","iopub.status.idle":"2024-12-05T12:04:02.642574Z","shell.execute_reply.started":"2024-12-05T12:04:02.625976Z","shell.execute_reply":"2024-12-05T12:04:02.641259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom torch.utils.data import Subset\n\n# Instantiate dataset and dataloader\ntrain_dataset = MosquitoDataset(\n    image_dir='/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/images',\n    label_dir='/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/labels',\n)\n\nindices = list(range(len(train_dataset)))\ntrain_indices, val_indices = train_test_split(indices, test_size=0.3, random_state=42)\n\ntrain_subset = Subset(train_dataset, train_indices)\nval_subset = Subset(train_dataset, val_indices)\n\ntrain_loader = DataLoader(train_subset, batch_size=2, shuffle=True, collate_fn=collate_fn)\nval_loader = DataLoader(val_subset, batch_size=2, shuffle=False, collate_fn=collate_fn)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:04:02.858563Z","iopub.execute_input":"2024-12-05T12:04:02.859098Z","iopub.status.idle":"2024-12-05T12:04:03.025071Z","shell.execute_reply.started":"2024-12-05T12:04:02.859001Z","shell.execute_reply":"2024-12-05T12:04:03.023603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model(model_name, num_classes):\n    \"\"\"\n    Fetches the specified model architecture and modifies the head for the given number of classes.\n\n    Args:\n        model_name (str): The name of the model architecture.\n        num_classes (int): The number of classes in the dataset.\n\n    Returns:\n        model: The configured PyTorch model.\n    \"\"\"\n    if model_name == \"fasterrcnn_resnet50\":\n        model = models.detection.fasterrcnn_resnet50_fpn(pretrained=True)\n        in_features = model.roi_heads.box_predictor.cls_score.in_features\n        model.roi_heads.box_predictor = models.detection.faster_rcnn.FastRCNNPredictor(in_features, num_classes)\n    \n    elif model_name == \"ssd300_vgg16\":\n        model = models.detection.ssdlite320_mobilenet_v3_large(pretrained=True)\n        # Extract the in_channels from the current classification head\n        in_channels = model.head.classification_head[0].in_channels\n        num_anchors = model.head.classification_head[0].out_channels // (len(model.head.classification_head) * num_classes)\n    \n        # Create a new classification head\n        new_classification_head = models.detection.ssdlite.SSDLiteClassificationHead(in_channels, num_anchors, num_classes)\n        model.head.classification_head = new_classification_head\n    \n    elif model_name == \"retinanet_resnet50\":\n        model = models.detection.retinanet_resnet50_fpn(pretrained=True)\n        in_features = model.head.classification_head.cls_logits.out_channels\n        model.head.classification_head.cls_logits = torch.nn.Conv2d(in_features, num_classes, kernel_size=3, padding=1)\n    \n    elif model_name == \"maskrcnn_resnet50\":\n        model = models.detection.maskrcnn_resnet50_fpn(pretrained=True)\n        in_features = model.roi_heads.box_predictor.cls_score.in_features\n        model.roi_heads.box_predictor = models.detection.faster_rcnn.FastRCNNPredictor(in_features, num_classes)\n        model.roi_heads.mask_predictor = None\n    \n    elif model_name == \"keypointrcnn_resnet50\":\n        model = models.detection.keypointrcnn_resnet50_fpn(pretrained=True)\n        in_features = model.roi_heads.keypoint_predictor.kps_score_lowres.in_channels\n        model.roi_heads.keypoint_predictor.kps_score_lowres = torch.nn.ConvTranspose2d(\n            in_features, num_classes * 2, kernel_size=4, stride=2, padding=1\n        )\n    \n    else:\n        raise ValueError(f\"Unsupported model name: {model_name}\")\n    \n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:08:31.244737Z","iopub.execute_input":"2024-12-05T12:08:31.246316Z","iopub.status.idle":"2024-12-05T12:08:31.260656Z","shell.execute_reply.started":"2024-12-05T12:08:31.246244Z","shell.execute_reply":"2024-12-05T12:08:31.258515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to evaluate the model\ndef evaluate_model(model, dataloader, device):\n    model.eval()\n    all_preds = []\n    all_targets = []\n    losses = []\n    \n    with torch.no_grad():\n        for images, targets in tqdm(dataloader):\n            images = [image.to(device) for image in images]\n            image_ids = [{'image_id': t['image_id']} for t in targets]\n            targets = [{k: v.to(device) for k, v in t.items() if k != 'image_id'} for t in targets]\n\n             # Temporarily enable training mode to calculate loss\n            model.train()\n            loss_dict = model(images, targets)\n            model.eval()  # Switch back to eval mode after loss computation\n            \n            # Sum losses from loss_dict\n            loss = sum(loss_dict.values())\n            losses.append(loss.item())\n\n            # Collect predictions\n            detections = model(images)  # Outputs predictions\n            for i, detection in enumerate(detections):\n                detection['image_id'] = image_ids[i]['image_id']\n                all_preds.append(detection)\n                all_targets.append({\n                    'boxes': targets[i]['boxes'].cpu().numpy(),\n                    'labels': targets[i]['labels'].cpu().numpy(),\n                    'image_id': image_ids[i]['image_id']\n                })\n            \n            # Collect predictions and targets for mAP calculation\n            all_preds.extend(detections)\n            all_targets.extend(targets)\n\n    avg_loss = sum(losses) / len(losses)\n    \n    # Calculate mAP (replace with custom implementation or a library function as needed)\n    mean_ap = calculate_mean_average_precision(all_preds, all_targets)\n\n    return avg_loss, mean_ap\n\n\n# Train and evaluate function\ndef train_and_evaluate_model(model, train_loader, val_loader, device, num_epochs=5):\n    model.train()\n    optimizer = torch.optim.Adam(model.parameters(), lr=1e-5)\n    train_loss_stats = []\n    val_loss_stats = []\n    map_stats = []\n\n    for epoch in range(num_epochs):\n        model.train()\n        epoch_loss = 0\n        \n        for images, targets in tqdm(train_loader):\n            images = [image.to(device) for image in images]\n            targets = [{k: v.to(device) for k, v in t.items() if k != 'image_id'} for t in targets]\n\n            # Zero the gradients\n            optimizer.zero_grad()\n            \n            # Forward pass\n            loss_dict = model(images, targets)\n            #print(loss_dict)\n            losses = sum(loss for loss in loss_dict.values())\n            epoch_loss += losses.item()\n            \n            # Backward pass\n            losses.backward()\n            optimizer.step()\n\n        # Average training loss\n        avg_train_loss = epoch_loss / len(train_loader)\n        train_loss_stats.append(avg_train_loss)\n\n        # Validation\n        avg_val_loss, mean_ap = evaluate_model(model, val_loader, device)\n        val_loss_stats.append(avg_val_loss)\n        map_stats.append(mean_ap)\n\n        # Print epoch stats\n        print(f\"Epoch {epoch+1}/{num_epochs}\")\n        print(f\"  Train Loss: {avg_train_loss:.4f}\")\n        print(f\"  Val Loss: {avg_val_loss:.4f}\")\n        print(f\"  mAP: {mean_ap:.4f}\")\n\n    return model, train_loss_stats, val_loss_stats, map_stats\n\n\n# Calculate mAP (Dummy Implementation)\ndef calculate_mean_average_precision(predictions, targets):\n    \"\"\"\n    Dummy implementation for mAP.\n    Replace this with your actual mAP calculation logic.\n    \"\"\"\n    # Placeholder: Use library or custom implementation for actual mAP\n    return 0.5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:04:03.311343Z","iopub.execute_input":"2024-12-05T12:04:03.311817Z","iopub.status.idle":"2024-12-05T12:04:03.332244Z","shell.execute_reply.started":"2024-12-05T12:04:03.311771Z","shell.execute_reply":"2024-12-05T12:04:03.330579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport pandas as pd\nfrom torchvision import models\nfrom torchvision.transforms import functional as F\nfrom torch.utils.data import DataLoader\n\n# Prepare test dataset and DataLoader\nclass TestDataset(Dataset):\n    def __init__(self, image_dir, transform=None):\n        self.image_dir = image_dir\n        self.transform = transform\n        self.image_ids = [f.split('.')[0] for f in os.listdir(image_dir)]\n\n    def __len__(self):\n        return len(self.image_ids)\n\n    def __getitem__(self, idx):\n        image_id = self.image_ids[idx]\n        image_path = os.path.join(self.image_dir, f\"{image_id}.jpeg\")\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n        if self.transform:\n            image = self.transform(image)\n        image = image.astype(np.float32)/255\n        return image, image_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:04:03.508756Z","iopub.execute_input":"2024-12-05T12:04:03.509360Z","iopub.status.idle":"2024-12-05T12:04:03.518956Z","shell.execute_reply.started":"2024-12-05T12:04:03.509314Z","shell.execute_reply":"2024-12-05T12:04:03.517433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\n# Function to perform prediction and save results\ndef predict_and_save(model, test_loader, device, output_csv_path):\n    model.eval()  # Set the model to evaluation mode\n    predictions = []\n\n    i = 1\n    with torch.no_grad():  # Disable gradient calculation during inference\n        for images, image_ids in test_loader:\n            \n            #print(images.shape)\n            images = [image.permute(2,0,1).to(device) for image in images]  # Send images to device\n            outputs = model(images)\n\n            c, h, w = images[0].shape\n            print(i, len(images), (h, w))\n            \n            i+=1\n            \n            # Process the predictions for each image\n            for i, output in enumerate(outputs):\n                boxes = output['boxes'].cpu().numpy()\n                labels = output['labels'].cpu().numpy()\n                scores = output['scores'].cpu().numpy()\n\n                #print(test_loader.dataset.image_ids[i], image_ids[0])\n                #print(output)\n                #break\n                if len(scores) > 0:\n                    # Select the prediction with the highest confidence\n                    best_idx = np.argmax(scores)\n                    xmin = boxes[best_idx][0]/w\n                    xmax = boxes[best_idx][2]/w\n                    ymin = boxes[best_idx][1]/h\n                    ymax = boxes[best_idx][3]/h\n                    \n                    xcenter = (xmin + xmax) / 2\n                    ycenter = (ymin + ymax) / 2\n                    width = xmax - xmin\n                    height = ymax - ymin\n\n                    print(labels[best_idx], get_label_name(labels[best_idx]))\n                    predictions.append({\n                        \"id\": len(predictions),\n                        \"ImageID\": image_ids[0],\n                        \"LabelName\": get_label_name(labels[best_idx]),  # Convert label ID to name\n                        \"Conf\": scores[best_idx],\n                        \"xcenter\": xcenter,\n                        \"ycenter\": ycenter,\n                        \"bbx_width\": width,\n                        \"bbx_height\": height,\n                    })\n                else:\n                    # Handle the case where no object is detected for the image\n                    predictions.append({\n                        \"id\": len(predictions),\n                        \"ImageID\": image_ids[0],\n                        \"LabelName\": get_label_name(0),  # Default label for no detection\n                        \"Conf\": 0.0,\n                        \"xcenter\": 0.0,\n                        \"ycenter\": 0.0,\n                        \"bbx_width\": 0.0,\n                        \"bbx_height\": 0.0,\n                    })\n\n                \n\n    # Convert predictions to a DataFrame and save to CSV\n    df = pd.DataFrame(predictions)\n    df.to_csv(output_csv_path, index=False)\n    print(f\"Predictions saved to {output_csv_path}\")\n\n# Helper function to map label index to class name\ndef get_label_name(label_index):\n    # Replace with actual mapping from label index to your class names\n    label_map = {\n        0: \"aegypti\",\n        1: \"albopictus\",  # Example class\n        2: \"anopheles\",  # Example class\n        3: \"culex\",\n        4: \"culiseta\",\n        5: \"japonicus/koreicus\"\n    }\n    return label_map.get(label_index, \"aegypti\")\n\n# Assuming you have a DataLoader for the test set (test_loader)\n# And that 'model' is already loaded and moved to the correct device\n\n\ntest_dataset = TestDataset(image_dir='/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images')\ntest_loader = DataLoader(test_dataset, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:04:04.227597Z","iopub.execute_input":"2024-12-05T12:04:04.228090Z","iopub.status.idle":"2024-12-05T12:04:04.298362Z","shell.execute_reply.started":"2024-12-05T12:04:04.228030Z","shell.execute_reply":"2024-12-05T12:04:04.296656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare for training\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel_names = ['fasterrcnn_resnet50']\nfor model_name in model_names:\n  # Load the model\n  model = get_model(model_name, num_classes=7)  # 6 classes + 1 for background\n  model = model.to(device)\n  # Train and evaluate\n  model, train_loss, val_loss, map_stats = train_and_evaluate_model(model, train_loader, val_loader, device, num_epochs=8)\n  torch.save(model, model_name+\".pth\")\n\n  # Run predictions and save them\n  output_csv_path = model_name+\" submission.csv\"\n  predict_and_save(model, test_loader, device, output_csv_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T12:07:08.932793Z","iopub.execute_input":"2024-12-05T12:07:08.933244Z","iopub.status.idle":"2024-12-05T12:07:15.762109Z","shell.execute_reply.started":"2024-12-05T12:07:08.933206Z","shell.execute_reply":"2024-12-05T12:07:15.759506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}