{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":85240,"databundleVersionId":9622164,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:21.747367Z","iopub.execute_input":"2024-12-06T08:36:21.748017Z","iopub.status.idle":"2024-12-06T08:36:22.851713Z","shell.execute_reply.started":"2024-12-06T08:36:21.747986Z","shell.execute_reply":"2024-12-06T08:36:22.850696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:22.853398Z","iopub.execute_input":"2024-12-06T08:36:22.853680Z","iopub.status.idle":"2024-12-06T08:36:25.897075Z","shell.execute_reply.started":"2024-12-06T08:36:22.853654Z","shell.execute_reply":"2024-12-06T08:36:25.896064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n!pip install ultralytics --upgrade \n\n!pip install -U ipywidgets -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:25.898345Z","iopub.execute_input":"2024-12-06T08:36:25.898732Z","iopub.status.idle":"2024-12-06T08:36:46.137058Z","shell.execute_reply.started":"2024-12-06T08:36:25.898704Z","shell.execute_reply":"2024-12-06T08:36:46.136048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nimport os\nimport pandas as pd\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:46.139346Z","iopub.execute_input":"2024-12-06T08:36:46.139633Z","iopub.status.idle":"2024-12-06T08:36:48.393155Z","shell.execute_reply.started":"2024-12-06T08:36:46.139606Z","shell.execute_reply":"2024-12-06T08:36:48.392400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data\"\nTRAIN_IMG_DIR = os.path.join(BASE_DIR, \"train/images\")\nTRAIN_LABEL_DIR = os.path.join(BASE_DIR, \"train/labels\")\nTEST_IMG_DIR = os.path.join(BASE_DIR, \"test/images\")\nOUTPUT_DIR = \"./runs/detect\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:48.394212Z","iopub.execute_input":"2024-12-06T08:36:48.394661Z","iopub.status.idle":"2024-12-06T08:36:48.399336Z","shell.execute_reply.started":"2024-12-06T08:36:48.394635Z","shell.execute_reply":"2024-12-06T08:36:48.398407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_content = f\"\"\"\npath: {BASE_DIR}  # Base directory for train/val/test\ntrain: train/images  # Path to training images (relative to 'path')\nval: train/images  # Path to validation images (relative to 'path')\ntest: test  # Path to test images (relative to 'path')\nnames:\n  0: aegypti\n  1: albopictus\n  2: anopheles\n  3: culex\n  4: culiseta\n  5: japonicus/koreicus\n\"\"\"\n\n\n# Save YAML configuration to a file\nwith open(\"data.yaml\", \"w\") as f:\n    f.write(yaml_content)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:48.400863Z","iopub.execute_input":"2024-12-06T08:36:48.401242Z","iopub.status.idle":"2024-12-06T08:36:48.412984Z","shell.execute_reply.started":"2024-12-06T08:36:48.401201Z","shell.execute_reply":"2024-12-06T08:36:48.412031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = YOLO(\"yolov8n.pt\")  # Using the smallest model for speed\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:36:48.414253Z","iopub.execute_input":"2024-12-06T08:36:48.414654Z","iopub.status.idle":"2024-12-06T08:36:48.985135Z","shell.execute_reply.started":"2024-12-06T08:36:48.414617Z","shell.execute_reply":"2024-12-06T08:36:48.983890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.train(\n    data=\"data.yaml\",  # Path to the YAML file\n    imgsz=640,  # Image size\n    epochs=50,  # Number of epochs\n    batch=16,  # Batch size\n    workers=20,  # Number of workers\n    device=device,  # Explicitly set the device (GPU or CPU)\n    project=OUTPUT_DIR,  # Output directory for results\n    name=\"yolov8_mosquito\",  # Experiment name\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:15:17.861821Z","iopub.execute_input":"2024-12-06T12:15:17.862760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom PIL import Image\n\n\n# List to store submission data\nsubmission_data = []\n\n# Loop through each test image\nfor idx, img_file in tqdm(enumerate(os.listdir(TEST_IMG_DIR)), total=len(os.listdir(TEST_IMG_DIR))):\n    img_path = os.path.join(TEST_IMG_DIR, img_file)\n    results = model(img_path)  # Perform inference on the image\n\n    # Remove the file extension\n    img_file_name = os.path.splitext(img_file)[0] + \".jpeg\"  # Ensure `.jpeg` extension\n\n    # Open the image to get its dimensions\n    with Image.open(img_path) as img:\n        img_width, img_height = img.size\n\n    # Initialize variables to store the best prediction\n    highest_conf = -1\n    best_prediction = None\n\n    # Loop through detections to find the one with the highest confidence\n    for box in result.boxes.data.tolist():  # Extract detections\n        x_center, y_center, width, height = box[:4]  # Absolute bounding box values\n        conf = box[4]  # Confidence score\n        cls = int(box[5])  # Class ID\n        class_name = model.names[cls]  # Class name (LabelName)\n    \n        # Ensure LabelName is clean\n        class_name = os.path.splitext(class_name)[0]\n    \n        # Normalize bounding box values\n        x_center /= img_width\n        y_center /= img_height\n        width /= img_width\n        height /= img_height\n    \n        # Check for highest confidence\n        if conf > highest_conf:\n            highest_conf = conf\n            best_prediction = [\n                idx,                     # id\n                img_file_name,           # ImageID\n                class_name,              # LabelName\n                conf,          # Conf (confidence score)\n                x_center,     # xcenter\n                y_center,      # ycenter\n                width,       # bbx_width\n                height      # bbx_height\n            ]\n\n\n    # Add the best prediction for this image (or a default if no detection)\n    if best_prediction is not None:\n        submission_data.append(best_prediction)\n    else:\n        # Handle case where no object is detected for the image\n        submission_data.append([idx, img_file_name, \"unknown\", 0.0, 0.0, 0.0, 0.0, 0.0])\n\n# Convert submission data to DataFrame\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:13:53.269067Z","iopub.status.idle":"2024-12-06T12:13:53.269627Z","shell.execute_reply.started":"2024-12-06T12:13:53.269370Z","shell.execute_reply":"2024-12-06T12:13:53.269395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns = [\"id\", \"ImageID\", \"LabelName\", \"Conf\", \"xcenter\", \"ycenter\", \"bbx_width\", \"bbx_height\"]\nsubmission_df = pd.DataFrame(submission_data, columns=columns)\n\n# Match the data types of all columns in final_submission to those in sample_submission\nfor col in sample_submission.columns:\n    final_submission[col] = final_submission[col].astype(sample_submission[col].dtype)\n\n# Save to CSV\nsubmission_output_path = \"/kaggle/working/submission.csv\"  # Update path as needed\nsubmission_df.to_csv(submission_output_path, index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:13:53.271039Z","iopub.status.idle":"2024-12-06T12:13:53.271427Z","shell.execute_reply.started":"2024-12-06T12:13:53.271260Z","shell.execute_reply":"2024-12-06T12:13:53.271276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T12:13:53.273417Z","iopub.status.idle":"2024-12-06T12:13:53.273734Z","shell.execute_reply.started":"2024-12-06T12:13:53.273588Z","shell.execute_reply":"2024-12-06T12:13:53.273604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}