{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":85240,"databundleVersionId":9622164,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:22:23.983727Z","iopub.execute_input":"2024-12-04T17:22:23.984236Z","iopub.status.idle":"2024-12-04T17:22:34.624140Z","shell.execute_reply.started":"2024-12-04T17:22:23.984178Z","shell.execute_reply":"2024-12-04T17:22:34.623196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nfrom ultralytics import YOLO\nfrom pathlib import Path\nimport cv2\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:22:34.625973Z","iopub.execute_input":"2024-12-04T17:22:34.626249Z","iopub.status.idle":"2024-12-04T17:22:39.988608Z","shell.execute_reply.started":"2024-12-04T17:22:34.626222Z","shell.execute_reply":"2024-12-04T17:22:39.987674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Original dataset paths\ntrain_images_dir = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/images\"\ntrain_labels_dir = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/train/labels\"\n\n# Create directories for the new train/validation split\noutput_dir = \"/kaggle/working/\"\ntrain_split_images = f\"{output_dir}/train_split/images\"\ntrain_split_labels = f\"{output_dir}/train_split/labels\"\nval_split_images = f\"{output_dir}/val_split/images\"\nval_split_labels = f\"{output_dir}/val_split/labels\"\n\n# Create necessary directories\nos.makedirs(train_split_images, exist_ok=True)\nos.makedirs(train_split_labels, exist_ok=True)\nos.makedirs(val_split_images, exist_ok=True)\nos.makedirs(val_split_labels, exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:22:39.989804Z","iopub.execute_input":"2024-12-04T17:22:39.990317Z","iopub.status.idle":"2024-12-04T17:22:39.996786Z","shell.execute_reply.started":"2024-12-04T17:22:39.990278Z","shell.execute_reply":"2024-12-04T17:22:39.995964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get all image and label file names\nimage_files = sorted(os.listdir(train_images_dir))\nlabel_files = sorted(os.listdir(train_labels_dir))\n\n# Ensure matching images and labels\nassert len(image_files) == len(label_files), \"Mismatch between images and labels\"\n\n# Split the data\ntrain_images, val_images, train_labels, val_labels = train_test_split(\n    image_files, label_files, test_size=0.2, random_state=42\n)\n\n# Copy files to new directories\ndef copy_files(file_list, src_dir, dest_dir):\n    for file_name in file_list:\n        shutil.copy(os.path.join(src_dir, file_name), os.path.join(dest_dir, file_name))\n\ncopy_files(train_images, train_images_dir, train_split_images)\ncopy_files(train_labels, train_labels_dir, train_split_labels)\ncopy_files(val_images, train_images_dir, val_split_images)\ncopy_files(val_labels, train_labels_dir, val_split_labels)\n\nprint(f\"Train images: {len(train_images)}, Validation images: {len(val_images)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:22:39.998890Z","iopub.execute_input":"2024-12-04T17:22:39.999149Z","iopub.status.idle":"2024-12-04T17:25:19.407326Z","shell.execute_reply.started":"2024-12-04T17:22:39.999125Z","shell.execute_reply":"2024-12-04T17:25:19.406399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_yaml = f\"\"\"\ntrain: {train_split_images}\nval: {val_split_images}\n\nnc: 6\nnames: ['aegypti', 'albopictus', 'anopheles', 'culex', 'culiseta', 'japonicus/koreicus']\n\"\"\"\n\nwith open(\"/kaggle/working/data.yaml\", \"w\") as f:\n    f.write(data_yaml)\n\nprint(\"Data configuration saved as data.yaml.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:25:19.408479Z","iopub.execute_input":"2024-12-04T17:25:19.408773Z","iopub.status.idle":"2024-12-04T17:25:19.414226Z","shell.execute_reply.started":"2024-12-04T17:25:19.408747Z","shell.execute_reply":"2024-12-04T17:25:19.413392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Load YOLO model (with pretrained weights for medium model)\nmodel = YOLO(\"yolov8m.pt\")  # Use yolov8l.pt for the large model\n\n# Train the model\nmodel.train(\n    data=\"/kaggle/working/data.yaml\",  # Path to data configuration\n    epochs=50,                        # Higher max epochs\n    imgsz=640,                         # Input image size\n    batch=32,                          # Larger batch size if GPU allows\n    name=\"mosquito_detection\",         # Experiment name\n    workers=4,                         # Increase for better data loading\n    device=0,                          # Use GPU\n    patience=5,                        # Early stopping patience\n    lr0=0.01,                          # Learning rate\n    lrf=0.1,                           # Final learning rate fraction\n    momentum=0.937,                    # Momentum\n    weight_decay=0.0005,               # L2 regularization\n    augment=True,                      # Data augmentation\n    pretrained=True                    # Start from pretrained weights\n)\n\n# Validate the model\nresults = model.val(data=\"/kaggle/working/data.yaml\")\nprint(\"Validation results:\", results)\n\n# Predict on test data\ntest_images_path = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images\"\npredictions = model.predict(\n    source=test_images_path,\n    conf=0.25,        # Confidence threshold\n    save_txt=True,    # Save predictions in text format\n    save_conf=True    # Save confidence scores\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T17:25:19.415436Z","iopub.execute_input":"2024-12-04T17:25:19.416157Z","iopub.status.idle":"2024-12-04T19:45:12.347822Z","shell.execute_reply.started":"2024-12-04T17:25:19.416131Z","shell.execute_reply":"2024-12-04T19:45:12.346564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Load YOLO model\n# model = YOLO(\"yolov8s.yaml\")  # Use yolov8m.yaml or yolov8l.yaml for larger models\n\n# # Train the model\n# model.train(\n#     data=\"/kaggle/working/data.yaml\",  # Path to data configuration\n#     epochs=70,                         # Adjust epochs based on available resources\n#     imgsz=640,                         # Input image size\n#     batch=16,                          # Batch size\n#     name=\"mosquito_detection\",         # Experiment name\n#     workers=2,                         # Data loaders\n#     device=0                           # Use GPU if available\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:45:12.349905Z","iopub.execute_input":"2024-12-04T19:45:12.350215Z","iopub.status.idle":"2024-12-04T19:45:12.355081Z","shell.execute_reply.started":"2024-12-04T19:45:12.350187Z","shell.execute_reply":"2024-12-04T19:45:12.354391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = model.val(data=\"/kaggle/working/data.yaml\")\nprint(\"Validation results:\", results)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:45:12.355981Z","iopub.execute_input":"2024-12-04T19:45:12.356266Z","iopub.status.idle":"2024-12-04T19:46:22.342065Z","shell.execute_reply.started":"2024-12-04T19:45:12.356222Z","shell.execute_reply":"2024-12-04T19:46:22.340904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images_path = \"/kaggle/input/dlp-object-detection/final_dlp_data/final_dlp_data/test/images\"\n\npredictions = model.predict(\n    source=test_images_path,\n    conf=0.25,        # Confidence threshold\n    save_txt=True,    # Save predictions in text format\n    save_conf=True    # Save confidence scores\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:46:22.344143Z","iopub.execute_input":"2024-12-04T19:46:22.344634Z","iopub.status.idle":"2024-12-04T19:49:20.320434Z","shell.execute_reply.started":"2024-12-04T19:46:22.344580Z","shell.execute_reply":"2024-12-04T19:49:20.319378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:49:20.327741Z","iopub.execute_input":"2024-12-04T19:49:20.330579Z","iopub.status.idle":"2024-12-04T19:49:20.339764Z","shell.execute_reply.started":"2024-12-04T19:49:20.330542Z","shell.execute_reply":"2024-12-04T19:49:20.338753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_boxes = sum(len(prediction.boxes) for prediction in predictions)\nprint(f\"Total bounding boxes: {total_boxes}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:49:20.340945Z","iopub.execute_input":"2024-12-04T19:49:20.341232Z","iopub.status.idle":"2024-12-04T19:49:20.359784Z","shell.execute_reply.started":"2024-12-04T19:49:20.341205Z","shell.execute_reply":"2024-12-04T19:49:20.358895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom PIL import Image\nimport random\n\n# Map class indices to names\nclass_names = ['aegypti', 'albopictus', 'anopheles', 'culex', 'culiseta', 'japonicus/koreicus']\n\nsubmission_data = []\n\n# Loop through all predictions\nfor prediction in predictions:\n    image_id = Path(prediction.path).name  # Extract image file name\n    \n    # Open the image to get its dimensions\n    image_path = prediction.path\n    with Image.open(image_path) as img:\n        img_width, img_height = img.size  # Get image dimensions\n    \n    # If no boxes are detected for this image, add a random class label\n    if len(prediction.boxes) == 0:\n        random_class = random.choice(class_names)  # Randomly select a class label\n        submission_data.append([\n            len(submission_data),  # ID\n            image_id,              # Image ID\n            random_class,          # Random class label\n            round(0.0, 1),         # No confidence score\n            round(0.0, 1),         # No bounding box center\n            round(0.0, 1),         # No bounding box center\n            round(0.0, 1),         # No bounding box dimensions\n            round(0.0, 1)          # No bounding box dimensions\n        ])\n    else:\n        # If there are detections, pick the one with the highest confidence\n        best_box = max(prediction.boxes, key=lambda box: box.conf)\n        \n        class_id = int(best_box.cls)\n        conf = round(float(best_box.conf), 1)  # Round confidence to one decimal\n\n        # Get the bounding box coordinates (x_center, y_center, width, height)\n        xywh = best_box.xywh.squeeze().tolist()\n        x_center, y_center, width, height = xywh\n        \n        # Normalize the coordinates based on image dimensions and round to one decimal place\n        x_center = round(x_center / img_width, 1)\n        y_center = round(y_center / img_height, 1)\n        width = round(width / img_width, 1)\n        height = round(height / img_height, 1)\n        \n        # Append the detection data\n        submission_data.append([\n            len(submission_data),  # ID\n            image_id,              # Image ID\n            class_names[class_id], # Class label\n            conf,                  # Confidence score (rounded)\n            x_center,              # x_center (normalized and rounded)\n            y_center,              # y_center (normalized and rounded)\n            width,                 # width (normalized and rounded)\n            height                 # height (normalized and rounded)\n        ])\n\n# Convert to DataFrame and save as CSV\nsubmission_df = pd.DataFrame(\n    submission_data,\n    columns=[\"id\", \"ImageID\", \"LabelName\", \"Conf\", \"xcenter\", \"ycenter\", \"bbx_width\", \"bbx_height\"]\n)\n\n# Make sure no NaNs or unexpected values are present\nsubmission_df = submission_df.fillna(0.0)\n\n# Save as CSV\nsubmission_file = \"/kaggle/working/21f1005104.csv\"\nsubmission_df.to_csv(submission_file, index=False)\n\nprint(f\"Submission file saved: {submission_file}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:49:20.360901Z","iopub.execute_input":"2024-12-04T19:49:20.361297Z","iopub.status.idle":"2024-12-04T19:49:22.871211Z","shell.execute_reply.started":"2024-12-04T19:49:20.361258Z","shell.execute_reply":"2024-12-04T19:49:22.870143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T19:49:22.872298Z","iopub.execute_input":"2024-12-04T19:49:22.872617Z","iopub.status.idle":"2024-12-04T19:49:22.878196Z","shell.execute_reply.started":"2024-12-04T19:49:22.872589Z","shell.execute_reply":"2024-12-04T19:49:22.877367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}