{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":85240,"databundleVersionId":9622164,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q ultralytics\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:45:46.315760Z","iopub.execute_input":"2024-12-06T12:45:46.316470Z","iopub.status.idle":"2024-12-06T12:45:58.321430Z","shell.execute_reply.started":"2024-12-06T12:45:46.316438Z","shell.execute_reply":"2024-12-06T12:45:58.320480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport os\nfrom ultralytics import YOLO\n\n# Check if GPU is available\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\n\n# Clear GPU Cache\ntorch.cuda.empty_cache()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:45:58.323163Z","iopub.execute_input":"2024-12-06T12:45:58.323448Z","iopub.status.idle":"2024-12-06T12:46:05.027461Z","shell.execute_reply.started":"2024-12-06T12:45:58.323420Z","shell.execute_reply":"2024-12-06T12:46:05.026529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# Define source and destination paths\nsource_dir = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train\"\ndest_dir = \"/kaggle/working/train\"\n\n# Copy images and labels to a writable location\nshutil.copytree(os.path.join(source_dir, \"images\"), os.path.join(dest_dir, \"images\"), dirs_exist_ok=True)\nshutil.copytree(os.path.join(source_dir, \"labels\"), os.path.join(dest_dir, \"labels\"), dirs_exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:46:05.028810Z","iopub.execute_input":"2024-12-06T12:46:05.029697Z","iopub.status.idle":"2024-12-06T12:50:01.668110Z","shell.execute_reply.started":"2024-12-06T12:46:05.029653Z","shell.execute_reply":"2024-12-06T12:50:01.667111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport os\nimport shutil\n\n# Paths to writable dataset\ntrain_images_dir = \"/kaggle/working/train/images\"\ntrain_labels_dir = \"/kaggle/working/train/labels\"\nval_images_dir = \"/kaggle/working/val/images\"\nval_labels_dir = \"/kaggle/working/val/labels\"\n\n# Create validation directories\nos.makedirs(val_images_dir, exist_ok=True)\nos.makedirs(val_labels_dir, exist_ok=True)\n\n# Get all image filenames\nimage_files = sorted([f for f in os.listdir(train_images_dir) if f.endswith((\".jpg\", \".jpeg\", \".png\"))])\n\n# Split into train and validation (80-20 split)\ntrain_files, val_files = train_test_split(image_files, test_size=0.2, random_state=42)\n\n# Copy validation files to validation folder\nfor file in val_files:\n    # Copy image\n    shutil.copy(os.path.join(train_images_dir, file), os.path.join(val_images_dir, file))\n    \n    # Copy corresponding label\n    label_file = file.replace(\".jpg\", \".txt\").replace(\".jpeg\", \".txt\").replace(\".png\", \".txt\")\n    shutil.copy(os.path.join(train_labels_dir, label_file), os.path.join(val_labels_dir, label_file))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:50:01.669755Z","iopub.execute_input":"2024-12-06T12:50:01.670059Z","iopub.status.idle":"2024-12-06T12:50:03.755112Z","shell.execute_reply.started":"2024-12-06T12:50:01.670034Z","shell.execute_reply":"2024-12-06T12:50:03.754113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_yaml = \"\"\"\ntrain: /kaggle/working/train/images\nval: /kaggle/working/val/images\n\nnc: 6\nnames: [\"aegypti\", \"albopictus\", \"anopheles\", \"culex\", \"culiseta\", \"japonicus/koreicus\"]\n\"\"\"\n\n# Save the YAML file\nwith open(\"mosquito_dataset.yaml\", \"w\") as f:\n    f.write(dataset_yaml)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:50:03.756534Z","iopub.execute_input":"2024-12-06T12:50:03.757119Z","iopub.status.idle":"2024-12-06T12:50:03.762528Z","shell.execute_reply.started":"2024-12-06T12:50:03.757078Z","shell.execute_reply":"2024-12-06T12:50:03.761833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the YOLOv5 model\nmodel = YOLO(\"yolov8n.pt\") # Use a pre-trained YOLOv8n model\n\n# Train the model\nmodel.train(\n    data=\"mosquito_dataset.yaml\",  # Path to the YAML file\n    epochs=20,                    # Number of epochs\n    imgsz=640,                    # Image size\n    batch=16,                     # Batch size\n    device=0,                     # GPU (0 for first GPU)\n    project=\"mosquito_detection\", # Project name\n    name=\"exp\"                    # Experiment name\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:50:03.763580Z","iopub.execute_input":"2024-12-06T12:50:03.763899Z","iopub.status.idle":"2024-12-06T13:53:00.830378Z","shell.execute_reply.started":"2024-12-06T12:50:03.763870Z","shell.execute_reply":"2024-12-06T13:53:00.829165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the best model from training\nbest_model_path = \"./mosquito_detection/exp/weights/best.pt\"\nmodel = YOLO(best_model_path)\n\n# Run inference on test images\nresults = model.predict(\n    source=\"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images\",  # Path to the test images\n    imgsz=640,              # Image size\n    conf=0.05,              # Confidence threshold\n    save_txt=True,          # Save predictions as .txt files\n    save=True               # Save annotated images\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T14:01:47.171165Z","iopub.execute_input":"2024-12-06T14:01:47.171524Z","iopub.status.idle":"2024-12-06T14:02:38.919256Z","shell.execute_reply.started":"2024-12-06T14:01:47.171493Z","shell.execute_reply":"2024-12-06T14:02:38.918472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T14:02:56.340463Z","iopub.execute_input":"2024-12-06T14:02:56.341250Z","iopub.status.idle":"2024-12-06T14:02:56.346974Z","shell.execute_reply.started":"2024-12-06T14:02:56.341214Z","shell.execute_reply":"2024-12-06T14:02:56.345833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results[0].boxes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T14:02:55.524350Z","iopub.execute_input":"2024-12-06T14:02:55.524603Z","iopub.status.idle":"2024-12-06T14:02:55.538359Z","shell.execute_reply.started":"2024-12-06T14:02:55.524580Z","shell.execute_reply":"2024-12-06T14:02:55.537402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport random\n\n# Define the class mapping for label names\nclass_mapping = {\n    0: \"aegypti\",\n    1: \"albopictus\",\n    2: \"anopheles\",\n    3: \"culex\",\n    4: \"culiseta\",\n    5: \"japonicus/koreicus\"\n}\n\n# Initialize an empty list to store the rows\ndata = []\n\n# Function to generate random label and confidence\ndef get_random_label_and_conf():\n    random_class_idx = random.choice(list(class_mapping.keys()))  # Random class index\n    random_conf = round(random.uniform(0.1, 0.9), 1)  # Random confidence between 0.1 and 0.9\n    label_name = class_mapping[random_class_idx]  # Get label name from random index\n    return label_name, random_conf\n\n# Iterate through the results (assuming `results` is the list of inference results from YOLO)\nfor result in results:\n    # Extract the image filename (ImageID)\n    image_id = result.path.split(\"/\")[-1].replace(\".jpeg\", \"\")  # Remove `.jpeg` from ImageID\n    \n    # Check if any bounding boxes are detected for the current image\n    if len(result.boxes) > 0:\n        # Initialize variables for the best detection (if needed)\n        max_confidence = 0.0\n        best_detection = None\n        \n        # Iterate over each detected box in the image\n        for box in result.boxes:\n            cls = int(box.cls)  # Class index\n            conf = box.conf.item()  # Confidence score\n            label_name = class_mapping[cls]  # Get label name from class index\n            xcenter, ycenter, width, height = box.xywhn[0].tolist()  # Normalized bounding box coordinates\n            \n            # Choose the detection with the highest confidence (or any other criteria)\n            if conf > max_confidence:\n                max_confidence = conf\n                best_detection = {\n                    \"LabelName\": label_name,\n                    \"Conf\": round(conf, 1),\n                    \"xcenter\": round(xcenter, 1),\n                    \"ycenter\": round(ycenter, 1),\n                    \"bbx_width\": round(width, 1),\n                    \"bbx_height\": round(height, 1)\n                }\n        \n        # Add the best detection row for this image\n        if best_detection:\n            data.append({\n                \"id\": len(data),\n                \"ImageID\": image_id,\n                \"LabelName\": best_detection[\"LabelName\"],\n                \"Conf\": best_detection[\"Conf\"],\n                \"xcenter\": best_detection[\"xcenter\"],\n                \"ycenter\": best_detection[\"ycenter\"],\n                \"bbx_width\": best_detection[\"bbx_width\"],\n                \"bbx_height\": best_detection[\"bbx_height\"]\n            })\n    else:\n        # If no bounding boxes are detected, assign a random label and confidence\n        label_name, random_conf = get_random_label_and_conf()\n        data.append({\n            \"id\": len(data),\n            \"ImageID\": image_id,\n            \"LabelName\": label_name,\n            \"Conf\": random_conf,\n            \"xcenter\": 0.0,  # Placeholder value for no bounding box\n            \"ycenter\": 0.0,  # Placeholder value for no bounding box\n            \"bbx_width\": 0.0,  # Placeholder value for no bounding box\n            \"bbx_height\": 0.0  # Placeholder value for no bounding box\n        })\n\n# Convert to DataFrame\ndf = pd.DataFrame(data)\n\n# Ensure the CSV output has exactly 525 rows (if there are fewer, pad with defaults)\nwhile len(df) < 525:\n    label_name, random_conf = get_random_label_and_conf()\n    df = df.append({\n        \"id\": len(df),\n        \"ImageID\": \"Unknown\",\n        \"LabelName\": label_name,\n        \"Conf\": random_conf,\n        \"xcenter\": 0.0,\n        \"ycenter\": 0.0,\n        \"bbx_width\": 0.0,\n        \"bbx_height\": 0.0\n    }, ignore_index=True)\n\n# Save to CSV\ndf.to_csv('predictions.csv', index=False)\n\n# Display the DataFrame\ndf\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T14:00:30.317953Z","iopub.execute_input":"2024-12-06T14:00:30.318267Z","iopub.status.idle":"2024-12-06T14:00:30.923815Z","shell.execute_reply.started":"2024-12-06T14:00:30.318238Z","shell.execute_reply":"2024-12-06T14:00:30.922940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}