{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":85240,"databundleVersionId":9622164,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:34.600781Z","iopub.execute_input":"2024-12-06T16:59:34.601097Z","iopub.status.idle":"2024-12-06T16:59:35.645379Z","shell.execute_reply.started":"2024-12-06T16:59:34.601065Z","shell.execute_reply":"2024-12-06T16:59:35.644406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install  ultralytics opencv-python matplotlib","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:35.646764Z","iopub.execute_input":"2024-12-06T16:59:35.647136Z","iopub.status.idle":"2024-12-06T16:59:45.918824Z","shell.execute_reply.started":"2024-12-06T16:59:35.647109Z","shell.execute_reply":"2024-12-06T16:59:45.917979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile data.yaml\npath: '/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data'\ntrain: '/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/images'\nval: '/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images'\n\n# class names\nnc: 6\nnames:\n  0: \"aegypti\"\n  1: \"albopictus\"\n  2: \"anopheles\"\n  3: \"culex\"\n  4: \"culiseta\"\n  5: \"japonicus/koreicus\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:45.920084Z","iopub.execute_input":"2024-12-06T16:59:45.920448Z","iopub.status.idle":"2024-12-06T16:59:45.926693Z","shell.execute_reply.started":"2024-12-06T16:59:45.920357Z","shell.execute_reply":"2024-12-06T16:59:45.925810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 50\nBATCH=16","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:45.928997Z","iopub.execute_input":"2024-12-06T16:59:45.929540Z","iopub.status.idle":"2024-12-06T16:59:45.937005Z","shell.execute_reply.started":"2024-12-06T16:59:45.929502Z","shell.execute_reply":"2024-12-06T16:59:45.936155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model.train(data='data.yaml', epochs=50, imgsz=640, batch=16)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:45.938097Z","iopub.execute_input":"2024-12-06T16:59:45.938379Z","iopub.status.idle":"2024-12-06T16:59:45.947906Z","shell.execute_reply.started":"2024-12-06T16:59:45.938354Z","shell.execute_reply":"2024-12-06T16:59:45.947143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport random\nimport pandas as pd\nimport numpy as np\nfrom PIL import Image\nimport torch\nfrom ultralytics import YOLO\n\nclass ObjectDetection:\n    dict1 = {\n        0: \"aegypti\",\n        1: \"albopictus\",\n        2: \"anopheles\",\n        3: \"culex\",\n        4: \"culiseta\",\n        5: \"japonicus/koreicus\"\n    }\n\n    def __init__(self):\n        self.device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n        print(\"Using Device: \", self.device)\n        self.model = self.load_model()\n    \n    def load_model(self):\n        model = YOLO()  \n        model.fuse() \n        model.train(data='data.yaml', epochs=50, imgsz=640, batch=16)\n        print(\"Model training complete.\")\n        return model\n\n    def predict(self, image):\n        results = self.model(image)\n        return results\n\n    def process_bboxes(self, results, image_id, image_name, image_height, image_width):\n        detections = []\n        valid_classes = list(self.dict1.values())  # List of valid class names\n        \n        for result in results:\n            boxes = result.boxes.cpu().numpy() if result.boxes is not None else []\n            \n            if len(boxes) == 0:  # No detections\n                random_label = random.choice(valid_classes)\n                detections.append({\n                    'id': image_id,\n                    'ImageID': image_name,\n                    'LabelName': random_label,\n                    'Conf': np.round(random.random(), 3),\n                    'xcenter': np.round(random.random(), 3),\n                    'ycenter': np.round(random.random(), 3),\n                    'bbx_width': np.round(random.random(), 3),\n                    'bbx_height': np.round(random.random(), 3)\n                })\n            else:\n                for box in boxes:\n                    cls = int(box.cls[0])\n                    conf = box.conf[0]\n                    xmin, ymin, xmax, ymax = box.xyxy[0]\n                    xmid = (xmin + xmax) / 2 / image_width\n                    ymid = (ymin + ymax) / 2 / image_height\n                    width = (xmax - xmin) / image_width\n                    height = (ymax - ymin) / image_height\n                    \n                    # Assign a valid class if cls is not in dict1\n                    label_name = self.dict1.get(cls, random.choice(valid_classes))\n                    \n                    detections.append({\n                        'id': image_id,\n                        'ImageID': image_name,\n                        'LabelName': label_name,\n                        'Conf': np.round(conf, 3),\n                        'xcenter': np.round(xmid, 3),\n                        'ycenter': np.round(ymid, 3),\n                        'bbx_width': np.round(width, 3),\n                        'bbx_height': np.round(height, 3)\n                    })\n        return detections\n\ndef main():\n    detector = ObjectDetection()\n    \n    # Initialize a list to store all detections\n    all_detections = []\n    \n    # Load all JPEG images from the test folder\n    images = os.listdir(\"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images/\")\n    \n    # Process each image exactly once\n    for image_id, image_name in enumerate(sorted(images)):\n        if not image_name.endswith('.jpeg'):\n            continue  # Skip non-JPEG files\n        \n        print(f\"Processing {image_name}\")\n        \n        # Load the image and convert it to a numpy array\n        image_path = f\"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images/{image_name}\"\n        image = Image.open(image_path)\n        image = np.array(image)\n        image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)  # Convert to BGR format\n        \n        # Perform inference\n        results = detector.predict(image)\n        \n        # Process bounding boxes, passing image height and width\n        image_height, image_width = image.shape[:2]\n        detections = detector.process_bboxes(results, image_id, image_name, image_height, image_width)\n        \n        # Ensure only one detection per image if no objects are detected\n        if len(detections) == 0:\n            random_label = random.choice(list(detector.dict1.values()))\n            detections.append({\n                'id': image_id,\n                'ImageID': image_name,\n                'LabelName': random_label,\n                'Conf': np.round(random.random(), 3),\n                'xcenter': np.round(random.random(), 3),\n                'ycenter': np.round(random.random(), 3),\n                'bbx_width': np.round(random.random(), 3),\n                'bbx_height': np.round(random.random(), 3)\n            })\n        \n        # Append the detections for the current image\n        all_detections.extend(detections)\n    \n    # Convert detections list to a DataFrame\n    df = pd.DataFrame(all_detections)\n\n    df_unique = df.drop_duplicates(subset=['ImageID'], keep='first')\n    \n    # Ensure the number of rows matches the number of test images\n    if len(df_unique) != 525:\n        print(f\"Warning: Expected 525 rows, found {len(df_unique)}. Adjusting to match the test set.\")\n\n    # Save the unique predictions to CSV\n    df_unique.to_csv(\"submission.csv\", index=False)\n    print(\"Final submission file created: submission.csv\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T16:59:45.949199Z","iopub.execute_input":"2024-12-06T16:59:45.949539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oup=pd.read_csv(\"/kaggle/working/submission.csv\")\nprint(len(oup))\noup","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import shutil\n# import random\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Define paths\n# input_images_path = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/images\"\n# working_images_path = \"/kaggle/working/train_images\"  # New writable directory\n# os.makedirs(working_images_path, exist_ok=True)\n\n# # Copy images to the new directory\n# for image in os.listdir(input_images_path):\n#     shutil.copy(os.path.join(input_images_path, image), os.path.join(working_images_path, image))\n\n# print(f\"Copied images to: {working_images_path}\")\n\n# # Define paths for train and validation sets\n# val_images_path = \"/kaggle/working/val_images\"\n# os.makedirs(val_images_path, exist_ok=True)\n\n# # Parameters\n# val_split_ratio = 0.2  # 20% for validation\n\n# # Get all image filenames\n# all_images = os.listdir(working_images_path)\n\n# # Shuffle and split the dataset\n# random.shuffle(all_images)\n# num_val = int(len(all_images) * val_split_ratio)\n# val_images = all_images[:num_val]\n\n# # Move validation images to the new directory\n# for image in val_images:\n#     shutil.move(os.path.join(working_images_path, image), os.path.join(val_images_path, image))\n\n# print(f\"Moved {num_val} images to validation set.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Define the new YAML content\n# yaml_content = \"\"\"# dataset.yaml\n# train: /kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/images\n# val: /kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/val/images  # If applicable\n# test: /kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images\n# nc: 6\n# names: [\"aegypti\", \"albopictus\", \"anopheles\", \"culex\", \"culiseta\", \"japonicus/koreicus\"]\n# \"\"\"\n# yaml_file_path = \"/kaggle/working/dataset.yaml\"\n\n# # Write the updated content to the YAML file\n# with open(\"/kaggle/working/dataset.yaml\", 'w') as file:\n#     file.write(yaml_content)\n\n# print(\"YAML file updated with separate train and validation paths.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n# import os\n\n# Set paths for training and validation images\n# train_images = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/images\"  # Update to your Kaggle dataset path\n# train_labels = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/labels\"  # Update to your Kaggle dataset path\n# test_images = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test\"    # Update to your Kaggle dataset path\n\n# Define number of classes and their names\n# num_classes = 6\n# class_names = [\n#     \"aegypti\", \n#     \"albopictus\", \n#     \"anopheles\", \n#     \"culex\", \n#     \"culiseta\", \n#     \"japonicus/koreicus\"\n# ]\n\n# Initialize YOLO model\n# model = YOLO(\"yolov8n.pt\")  # Using the YOLOv8 nano model (lightweight)\n\n# # Train the model\n# model.train(\n#     # data={\n#     #     \"train\": train_images, \n#     #     \"val\": train_images,\n#     #     \"nc\": num_classes, \n#     #     \"names\": class_names\n#     # }, \n#     data=yaml_file_path,\n#     epochs=1,           # Adjust epochs based on your needs\n#     imgsz=640,           # Image size\n#     batch=16,            # Batch size\n#     workers=2,           # Number of data loader workers\n#     #device=0,            # Use GPU if available\n#     name=\"mosquito_detection\",  # Experiment name\n# )\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the trained model\n# model = YOLO(\"/kaggle/working/mosquito_detection/weights/best.pt\")  # Path to the best weights\n\n# # Perform inference on test images\n# results = model.predict(\n#     source=test_images,  # Path to test images\n#     conf=0.25,           # Confidence threshold\n#     save=True,           # Save predictions\n#     save_txt=True        # Save predictions as text files\n# )\n\n# # Print results\n# print(\"Inference complete. Check the output folder for predictions.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n\n# submission_data = []\n\n# for result in results:\n#     image_id = result.path.split(\"/\")[-1]\n#     for box in result.boxes:\n#         label_name = model.names[int(box.cls)]\n#         conf = box.conf.item()\n#         xcenter, ycenter, width, height = box.xywhn.tolist()[0]\n        \n#         submission_data.append([image_id, label_name, conf, xcenter, ycenter, width, height])\n\n# # Convert to DataFrame\n# submission_df = pd.DataFrame(\n#     submission_data,\n#     columns=[\"ImageID\", \"LabelName\", \"Conf\", \"xcenter\", \"ycenter\", \"bbx_width\", \"bbx_height\"]\n# )\n\n# # Add an ID column\n# submission_df.insert(0, \"id\", range(len(submission_df)))\n\n# # Save submission file\n# submission_df.to_csv(\"submission.csv\", index=False)\n# print(\"Submission file created: 'submission.csv'\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import pandas as pd\n# import numpy as np\n# import cv2\n# import torch\n# import torchvision.transforms as transforms\n# from torch.utils.data import Dataset, DataLoader\n# from torchvision.models.detection import fasterrcnn_resnet50_fpn\n# from torchvision.models.detection.faster_rcnn import FastRCNNPredictor\n\n# # Constants\n# DATA_DIR = '/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/'\n# TRAIN_IMAGES_DIR = os.path.join(DATA_DIR, 'train/images')\n# TRAIN_LABELS_DIR = os.path.join(DATA_DIR, 'train/labels')\n# TEST_IMAGES_DIR = os.path.join(DATA_DIR, 'test/images')\n# SUBMISSION_FILE = 'submission.csv'\n\n# # Class mapping\n# CLASS_NAMES = [\"aegypti\", \"albopictus\", \"anopheles\", \"culex\", \"culiseta\", \"japonicus/koreicus\"]\n\n# # Custom Dataset Class\n# class MosquitoDataset(Dataset):\n#     def __init__(self, images_dir, labels_dir, transform=None):\n#         self.images_dir = images_dir\n#         self.labels_dir = labels_dir\n#         self.transform = transform\n#         self.images = os.listdir(images_dir)\n\n#     def __len__(self):\n#         return len(self.images)\n\n#     def __getitem__(self, idx):\n#         img_name = self.images[idx]\n#         img_path = os.path.join(self.images_dir, img_name)\n#         image = cv2.imread(img_path)\n#         image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n#         # Load labels\n#         label_path = os.path.join(self.labels_dir, img_name.replace('.jpg', '.txt'))\n#         boxes = []\n#         labels = []\n        \n#         if os.path.exists(label_path):\n#             with open(label_path, 'r') as f:\n#                 for line in f.readlines():\n#                     parts = line.strip().split()\n#                     if len(parts) < 5:  # Ensure there are enough parts\n#                         continue\n#                     class_label = int(parts[0])\n#                     x_center = float(parts[1])\n#                     y_center = float(parts[2])\n#                     width = float(parts[3])\n#                     height = float(parts[4])\n                    \n#                     # Convert to bounding box format\n#                     x1 = (x_center - width / 2) * image.shape[1]\n#                     y1 = (y_center - height / 2) * image.shape[0]\n#                     x2 = (x_center + width / 2) * image.shape[1]\n#                     y2 = (y_center + height / 2) * image.shape[0]\n                    \n#                     boxes.append([x1, y1, x2, y2])\n#                     labels.append(class_label)\n\n#         # Create target dictionary\n#         if len(boxes) == 0:\n#             boxes = torch.empty((0, 4), dtype=torch.float32)  # Correct shape for no boxes\n#             labels = torch.empty((0,), dtype=torch.int64)      # Correct shape for no labels\n#         else:\n#             boxes = torch.as_tensor(boxes, dtype=torch.float32)\n#             labels = torch.as_tensor(labels, dtype=torch.int64)\n\n#         target = {\n#             \"boxes\": boxes,\n#             \"labels\": labels\n#         }\n\n#         if self.transform:\n#             image = self.transform(image)\n\n#         # Return a single target for the current image\n#         return image, target  # Return a single target instead of a list\n\n\n\n# # Transformations\n# transform = transforms.Compose([\n#     transforms.ToPILImage(),\n#     transforms.Resize((640, 640)),\n#     transforms.ToTensor(),\n# ])\n\n# # Load datasets\n# train_dataset = MosquitoDataset(TRAIN_IMAGES_DIR, TRAIN_LABELS_DIR, transform=transform)\n# train_loader = DataLoader(train_dataset, batch_size=8, shuffle=True, num_workers=4)\n\n# # Define the model\n# def get_model(num_classes):\n#     model = fasterrcnn_resnet50_fpn(pretrained=True)\n#     in_features = model.roi_heads.box_predictor.cls_score.in_features\n#     model.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n#     return model\n\n# # Training the model\n# device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')\n# model = get_model(len(CLASS_NAMES)).to(device)\n\n# # Optimizer\n# params = [p for p in model.parameters() if p.requires_grad]\n# optimizer = torch.optim.SGD(params, lr=0.005, momentum=0.9, weight_decay=0.0005)\n\n# # Training Loop with Debugging\n# num_epochs = 1\n# model.train()\n# for epoch in range(num_epochs):\n#     for images, targets in train_loader:\n#         images = [image.to(device) for image in images]\n        \n#         # Debugging: Print targets to inspect their structure\n#         print(\"Targets before processing:\", targets)\n        \n#         try:\n#             targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n#         except Exception as e:\n#             print(f\"Error processing targets: {e}\")\n#             continue  # Skip this batch if there's an error\n\n#         loss_dict = model(images, targets)\n#         losses = sum(loss for loss in loss_dict.values())\n\n#         optimizer.zero_grad()\n#         losses.backward()\n#         optimizer.step()\n\n#     print(f'Epoch [{epoch + 1}/{num_epochs}], Loss: {losses.item():.4f}')\n\n# # Prediction on test set\n# model.eval()\n# test_images = os.listdir(TEST_IMAGES_DIR)\n# submission_data = []\n\n# for img_name in test_images:\n#     img_path = os.path.join(TEST_IMAGES_DIR, img_name)\n#     image = cv2.imread(img_path)\n#     image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#     image_tensor = transform(image).unsqueeze(0).to(device)\n\n#     with torch.no_grad():\n#         predictions = model(image_tensor)\n\n#     boxes = predictions[0]['boxes'].cpu().numpy()\n#     labels = predictions[0]['labels'].cpu().numpy()\n#     scores = predictions[0]['scores'].cpu().numpy()\n\n#     for i in range(len(boxes)):\n#         if scores[i] > 0.5:  # Confidence threshold\n#             x_center = (boxes[i][0] + boxes[i][2]) / 2 / image.shape[1]\n#             y_center = (boxes[i][1] + boxes[i][3]) / 2 / image.shape[0]\n#             width = (boxes[i][2] - boxes[i][0]) / image.shape[1]\n#             height = (boxes[i][3] - boxes[i][1]) / image.shape[0]\n#             submission_data.append([img_name, CLASS_NAMES[labels[i]], scores[i], x_center, y_center, width, height])\n\n# # Create submission DataFrame\n# submission_df = pd.DataFrame(submission_data, columns=['ImageID', 'LabelName', 'Conf', 'xcenter', 'ycenter', 'bbx_width', 'bbx_height'])\n# submission_df['id'] = submission_df.index\n# submission_df = submission_df[['id', 'ImageID', 'LabelName', 'Conf', 'xcenter', 'ycenter', 'bbx_width', 'bbx_height']]\n\n# # Save submission file\n# submission_df.to_csv(SUBMISSION_FILE, index=False)\n# print(f'Submission file saved as {SUBMISSION_FILE}')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}