{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31703,"databundleVersionId":2871752,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport random\n\nimport pandas as pd\n\n\n\n# Define paths\n\nsource_dir = \"/kaggle/input/tensorflow-great-barrier-reef/train_images\"\n\nyolo_data_dir = \"/kaggle/working/yolo_data\"\n\n\n\n# YOLO subdirectories\n\ntrain_dir = os.path.join(yolo_data_dir, \"train/images\")\n\ntrain_label_dir = os.path.join(yolo_data_dir, \"train/labels\")\n\nval_dir = os.path.join(yolo_data_dir, \"val/images\")\n\nval_label_dir = os.path.join(yolo_data_dir, \"val/labels\")\n\ntest_dir = os.path.join(yolo_data_dir, \"test/images\")\n\ntest_label_dir = os.path.join(yolo_data_dir, \"test/labels\")\n\n\n\n# Create directories\n\nos.makedirs(train_dir, exist_ok=True)\n\nos.makedirs(train_label_dir, exist_ok=True)\n\nos.makedirs(val_dir, exist_ok=True)\n\nos.makedirs(val_label_dir, exist_ok=True)\n\nos.makedirs(test_dir, exist_ok=True)\n\nos.makedirs(test_label_dir, exist_ok=True)\n\n\n\n# Load annotations\n\nannotations_file = \"/kaggle/input/tensorflow-great-barrier-reef/train.csv\"\n\nannotations_df = pd.read_csv(annotations_file)\n\n\n\n# Helper function to create symbolic links for images and create YOLO annotation files\n\ndef link_images_and_labels(video_folder, dest_folder, label_dir, annotations_df):\n\n    video_path = os.path.join(source_dir, video_folder)\n\n    for image_name in os.listdir(video_path):\n\n        image_path = os.path.join(video_path, image_name)\n\n\n\n        # Create a symbolic link for the image in the destination directory\n\n        link_name = os.path.join(dest_folder, image_name)\n\n        if not os.path.exists(link_name):  # Check if link already exists\n\n            os.symlink(image_path, link_name)\n\n\n\n        # Save YOLO annotation files for this image if available\n\n        video_id = int(video_folder.split('_')[1])\n\n        frame_number = int(image_name.split('.')[0])\n\n        image_annotations = annotations_df[\n\n            (annotations_df[\"video_id\"] == video_id) & \n\n            (annotations_df[\"video_frame\"] == frame_number)\n\n        ]\n\n\n\n        # Convert annotations to YOLO format if they exist\n\n        os.makedirs(label_dir, exist_ok=True)\n\n        label_path = os.path.join(label_dir, f\"{image_name.split('.')[0]}.txt\")\n\n\n\n        with open(label_path, \"w\") as label_file:\n\n            if not image_annotations.empty:\n\n                for _, row in image_annotations.iterrows():\n\n                    annotations = eval(row[\"annotations\"])  # Convert string to list of dicts\n\n                    for annotation in annotations:\n\n                        x = annotation[\"x\"]\n\n                        y = annotation[\"y\"]\n\n                        w = annotation[\"width\"]\n\n                        h = annotation[\"height\"]\n\n\n\n                        # Calculate YOLO format values\n\n                        x_center = (x + w / 2) / 1280  # Assuming 1280 width\n\n                        y_center = (y + h / 2) / 720   # Assuming 720 height\n\n                        width = w / 1280\n\n                        height = h / 720\n\n\n\n                        label_file.write(f\"0 {x_center} {y_center} {width} {height}\\n\")\n\n            else:\n\n                # Create a default label for background images\n\n                label_file.write(\"0 0.5 0.5 1 1\\n\")  # Class 0 (background), centered and full size\n\n\n\n# Link images for training\n\nlink_images_and_labels(\"video_0\", train_dir, train_label_dir, annotations_df)\n\nlink_images_and_labels(\"video_1\", train_dir, train_label_dir, annotations_df)\n\n\n\n# Split video_2 images equally for validation and test sets\n\nvideo_2_path = os.path.join(source_dir, \"video_2\")\n\nvideo_2_images = os.listdir(video_2_path)\n\nrandom.shuffle(video_2_images)\n\nsplit_idx = len(video_2_images) // 2\n\n\n\n# Link validation images and labels\n\nlink_images_and_labels(\"video_2\", val_dir, val_label_dir, annotations_df)\n\n\n\n# Link test images and labels\n\nfor image_name in video_2_images[split_idx:]:\n\n    image_path = os.path.join(video_2_path, image_name)\n\n    link_name = os.path.join(test_dir, image_name)\n\n    if not os.path.exists(link_name):\n\n        os.symlink(image_path, link_name)\n\n\n\n# Create empty label files for test images\n\nfor image_name in video_2_images[split_idx:]:\n\n    label_path = os.path.join(test_label_dir, f\"{image_name.split('.')[0]}.txt\")\n\n    if not os.path.exists(label_path):\n\n        with open(label_path, \"w\") as label_file:\n\n            label_file.write(\"\")  # Optional: can leave it empty or specify a format\n\n\n\nprint(\"Dataset organization with symbolic links completed.\")\n","metadata":{"execution":{"iopub.execute_input":"2024-10-30T11:00:24.060555Z","iopub.status.busy":"2024-10-30T11:00:24.059666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n\n\ndef validate_dataset(yolo_data_dir):\n\n    # Define directories\n\n    train_image_dir = os.path.join(yolo_data_dir, \"train/images\")\n\n    train_label_dir = os.path.join(yolo_data_dir, \"train/labels\")\n\n    val_image_dir = os.path.join(yolo_data_dir, \"val/images\")\n\n    val_label_dir = os.path.join(yolo_data_dir, \"val/labels\")\n\n    test_image_dir = os.path.join(yolo_data_dir, \"test/images\")\n\n    test_label_dir = os.path.join(yolo_data_dir, \"test/labels\")\n\n\n\n    # Function to count images and labels\n\n    def count_files(image_dir, label_dir):\n\n        image_files = os.listdir(image_dir)\n\n        label_files = os.listdir(label_dir)\n\n\n\n        num_images = len(image_files)\n\n        num_labels = len(label_files)\n\n        empty_labels = 0\n\n\n\n        for label_file in label_files:\n\n            label_path = os.path.join(label_dir, label_file)\n\n            with open(label_path, \"r\") as f:\n\n                content = f.read().strip()\n\n                if content == \"\":\n\n                    empty_labels += 1\n\n        \n\n        return num_images, num_labels, empty_labels\n\n\n\n    # Validate train set\n\n    train_counts = count_files(train_image_dir, train_label_dir)\n\n    print(\"Training set - Images: {}, Labels: {}, Empty Labels: {}\".format(*train_counts))\n\n\n\n    # Validate validation set\n\n    val_counts = count_files(val_image_dir, val_label_dir)\n\n    print(\"Validation set - Images: {}, Labels: {}, Empty Labels: {}\".format(*val_counts))\n\n\n\n    # Validate test set\n\n    test_counts = count_files(test_image_dir, test_label_dir)\n\n    print(\"Test set - Images: {}, Labels: {}, Empty Labels: {}\".format(*test_counts))\n\n\n\n# Run validation\n\nvalidate_dataset(yolo_data_dir)\n","metadata":{"execution":{"iopub.execute_input":"2024-10-30T11:20:10.516079Z","iopub.status.busy":"2024-10-30T11:20:10.51536Z","iopub.status.idle":"2024-10-30T11:20:11.265803Z","shell.execute_reply":"2024-10-30T11:20:11.264888Z","shell.execute_reply.started":"2024-10-30T11:20:10.516038Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clone the YOLOv5 repository\n\n!git clone https://github.com/ultralytics/yolov5.git\n\n%cd yolov5\n\n# Install dependencies\n\n!pip install -r requirements.txt\n","metadata":{"execution":{"iopub.execute_input":"2024-10-30T11:20:29.680724Z","iopub.status.busy":"2024-10-30T11:20:29.679943Z","iopub.status.idle":"2024-10-30T11:20:45.219689Z","shell.execute_reply":"2024-10-30T11:20:45.218591Z","shell.execute_reply.started":"2024-10-30T11:20:29.680683Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a YAML file for YOLOv5\n\nyaml_content = \"\"\"\n\ntrain: /kaggle/working/yolo_data/train/images\n\nval: /kaggle/working/yolo_data/val/images\n\ntest: /kaggle/working/yolo_data/test/images\n\n\n\nnc: 1  # Number of classes\n\nnames: ['COTS']  # Class names\n\n\"\"\"\n\n\n\n# Write to a data.yaml file\n\nwith open('/kaggle/working/yolo_data/data.yaml', 'w') as f:\n\n    f.write(yaml_content.strip())\n","metadata":{"execution":{"iopub.execute_input":"2024-10-30T11:20:45.222256Z","iopub.status.busy":"2024-10-30T11:20:45.22188Z","iopub.status.idle":"2024-10-30T11:20:45.229222Z","shell.execute_reply":"2024-10-30T11:20:45.228066Z","shell.execute_reply.started":"2024-10-30T11:20:45.222215Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import wandb\n\n\n\ntry:\n\n    from kaggle_secrets import UserSecretsClient\n\n    user_secrets = UserSecretsClient()\n\n    api_key = user_secrets.get_secret(\"WANDB\")\n\n    wandb.login(key=api_key)\n\n    anonymous = None\n\nexcept:\n\n    wandb.login(anonymous='must')\n\n    print('To use your W&B account,\\nGo to Add-ons -> Secrets and provide your W&B access token. Use the Label name as WANDB. \\nGet your W&B access token from here: https://wandb.ai/authorize')","metadata":{"execution":{"iopub.execute_input":"2024-10-30T11:20:49.257362Z","iopub.status.busy":"2024-10-30T11:20:49.256567Z","iopub.status.idle":"2024-10-30T11:20:52.798503Z","shell.execute_reply":"2024-10-30T11:20:52.797614Z","shell.execute_reply.started":"2024-10-30T11:20:49.257309Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import wandb\n\n\n\n# Start a new W&B run to track this script\n\nwandb.init(project=\"yolov5new\")\n\n\n\n!python train.py --img 1280 --batch 16 --epochs 30 --data /kaggle/working/yolo_data/data.yaml --weights yolov5s.pt --cache --project yolov5new\n\n\n\n\n\n# [Optional] finish the W&B run, necessary in notebooks\n\nwandb.finish()\n","metadata":{"execution":{"iopub.execute_input":"2024-10-30T11:20:58.12502Z","iopub.status.busy":"2024-10-30T11:20:58.124418Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n\n\n# Load the CSV file\n\ndf = pd.read_csv('/kaggle/input/crownofthorns-starfish/cots.csv')\n\n\n\n# Display the first five rows of the DataFrame\n\nprint(df.head())\n","metadata":{"execution":{"iopub.execute_input":"2024-10-27T10:45:14.152947Z","iopub.status.busy":"2024-10-27T10:45:14.152068Z","iopub.status.idle":"2024-10-27T10:45:14.189379Z","shell.execute_reply":"2024-10-27T10:45:14.188444Z","shell.execute_reply.started":"2024-10-27T10:45:14.152901Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save test.yaml for testing\n\nyaml_content = \"\"\"\n\n# test.yaml for testing the COTS detection model\n\n\n\n# Paths to your images\n\ntest: /kaggle/input/crownofthorns-starfish/images  # Path to test images\n\n\n\n# Dummy paths for train and val, pointing to your images (if necessary)\n\ntrain: /kaggle/input/crownofthorns-starfish/images  # Use the same images as a placeholder\n\nval: /kaggle/input/crownofthorns-starfish/images    # Use the same images as a placeholder\n\n\n\n# Number of classes\n\nnc: 1  # Update this if you have more classes\n\n\n\n# Class names\n\nnames: ['cots']  # Update with the actual class names\n\n\"\"\"\n\n\n\n# Write to file\n\nwith open('/kaggle/working/test.yaml', 'w') as f:\n\n    f.write(yaml_content)\n","metadata":{"execution":{"iopub.execute_input":"2024-10-27T10:56:32.967602Z","iopub.status.busy":"2024-10-27T10:56:32.96656Z","iopub.status.idle":"2024-10-27T10:56:32.973743Z","shell.execute_reply":"2024-10-27T10:56:32.972752Z","shell.execute_reply.started":"2024-10-27T10:56:32.967554Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python val.py --weights /kaggle/working/yolov5/yolov5new/exp2/weights/best.pt --data /kaggle/working/test.yaml --img 1280 --save-json\n","metadata":{"execution":{"iopub.execute_input":"2024-10-27T10:56:36.475132Z","iopub.status.busy":"2024-10-27T10:56:36.47418Z","iopub.status.idle":"2024-10-27T10:57:16.337117Z","shell.execute_reply":"2024-10-27T10:57:16.335904Z","shell.execute_reply.started":"2024-10-27T10:56:36.475089Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Replace 'exp' with your experiment folder name if different\nshutil.make_archive('/kaggle/working/validation_results', 'zip', '/kaggle/working/yolov5/runs/val/exp')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/working')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}