{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31703,"databundleVersionId":2871752,"sourceType":"competition"}],"dockerImageVersionId":30715,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Importing and loading Packages\n!git clone https://github.com/ultralytics/yolov5.git\n\n%cd yolov5\n!ls /kaggle/working/yolov5\n\n!pip install -r requirements.txt","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:03:45.253336Z","iopub.execute_input":"2024-06-05T18:03:45.253781Z","iopub.status.idle":"2024-06-05T18:04:05.011364Z","shell.execute_reply.started":"2024-06-05T18:03:45.253741Z","shell.execute_reply":"2024-06-05T18:04:05.009844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport ast\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\ntqdm.pandas()\nfrom shutil import copyfile\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:06.122068Z","iopub.execute_input":"2024-06-05T18:04:06.122556Z","iopub.status.idle":"2024-06-05T18:04:06.134944Z","shell.execute_reply.started":"2024-06-05T18:04:06.122517Z","shell.execute_reply":"2024-06-05T18:04:06.133451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Exploring the dataset**","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/tensorflow-great-barrier-reef/train.csv')\n\nprint(\"Training Data:\")\nprint(train_df.head())\nprint(train_df.columns)\n\ntrain_df['annotations'] = train_df['annotations'].apply(ast.literal_eval)\nnon_empty_annotations_df = train_df[train_df['annotations'].apply(len) > 0]\nprint(non_empty_annotations_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:08.105390Z","iopub.execute_input":"2024-06-05T18:04:08.105853Z","iopub.status.idle":"2024-06-05T18:04:08.738195Z","shell.execute_reply.started":"2024-06-05T18:04:08.105815Z","shell.execute_reply":"2024-06-05T18:04:08.736679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/tensorflow-great-barrier-reef/test.csv')\n\nprint(\"\\nTest Data:\")\nprint(test_df.head())\nprint(test_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:14.175523Z","iopub.execute_input":"2024-06-05T18:04:14.175983Z","iopub.status.idle":"2024-06-05T18:04:14.190418Z","shell.execute_reply.started":"2024-06-05T18:04:14.175936Z","shell.execute_reply":"2024-06-05T18:04:14.188770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of training images: ', len(train_df))\nprint('Number of test images: ', len(test_df))","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:16.704638Z","iopub.execute_input":"2024-06-05T18:04:16.705106Z","iopub.status.idle":"2024-06-05T18:04:16.712308Z","shell.execute_reply.started":"2024-06-05T18:04:16.705071Z","shell.execute_reply":"2024-06-05T18:04:16.710870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking each dataset\ndef data_indepth_check(df, name):\n    print(f\"\\Checking {name}:\")\n    print(df.info())\n    print(df.describe())\n    \n#train data check\ndata_indepth_check(train_df, \"Training Data\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:18.742742Z","iopub.execute_input":"2024-06-05T18:04:18.745366Z","iopub.status.idle":"2024-06-05T18:04:18.859710Z","shell.execute_reply.started":"2024-06-05T18:04:18.745142Z","shell.execute_reply":"2024-06-05T18:04:18.854774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test data check\ndata_indepth_check(test_df, \"Test Data\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:21.598655Z","iopub.execute_input":"2024-06-05T18:04:21.601675Z","iopub.status.idle":"2024-06-05T18:04:21.699685Z","shell.execute_reply.started":"2024-06-05T18:04:21.601483Z","shell.execute_reply":"2024-06-05T18:04:21.694536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing NAN values\nnan_summary = train_df.isnull().sum()\n\n\nprint(\"NaN values in training data:\")\nprint(nan_summary)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:24.442126Z","iopub.execute_input":"2024-06-05T18:04:24.445472Z","iopub.status.idle":"2024-06-05T18:04:24.489067Z","shell.execute_reply.started":"2024-06-05T18:04:24.445206Z","shell.execute_reply":"2024-06-05T18:04:24.483253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:26.471499Z","iopub.execute_input":"2024-06-05T18:04:26.473484Z","iopub.status.idle":"2024-06-05T18:04:26.511744Z","shell.execute_reply.started":"2024-06-05T18:04:26.473161Z","shell.execute_reply":"2024-06-05T18:04:26.506634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:04:28.238743Z","iopub.execute_input":"2024-06-05T18:04:28.243521Z","iopub.status.idle":"2024-06-05T18:04:28.285708Z","shell.execute_reply.started":"2024-06-05T18:04:28.243248Z","shell.execute_reply":"2024-06-05T18:04:28.279131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***YOLOv5 Implementation ***","metadata":{}},{"cell_type":"code","source":"FOLD = 1\nDIM = 3000\nMODEL = 'yolov5s'\nBATCH = 4\nEPOCHS = 10\nOPTIMIZER = 'SGD'\nIMG_SIZE = 256\nBATCH_SIZE = 16\nTRAIN_PATH = '/kaggle/input/tensorflow-great-barrier-reef/train_images'\nTEST_PATH = '/kaggle/input/tensorflow-great-barrier-reef/test_images'","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:03.238083Z","iopub.execute_input":"2024-06-05T18:07:03.238611Z","iopub.status.idle":"2024-06-05T18:07:03.246357Z","shell.execute_reply.started":"2024-06-05T18:07:03.238570Z","shell.execute_reply":"2024-06-05T18:07:03.244827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/tensorflow-great-barrier-reef/train.csv')\n\ntrain_df['path'] = [f\"{TRAIN_PATH}/video_{vid}/{frame}.jpg\" for vid, frame in zip(train_df['video_id'], train_df['video_frame'])]\ntrain_df['num_bbox'] = [len(ast.literal_eval(ann)) for ann in train_df['annotations']]","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:04.871078Z","iopub.execute_input":"2024-06-05T18:07:04.871566Z","iopub.status.idle":"2024-06-05T18:07:05.451114Z","shell.execute_reply.started":"2024-06-05T18:07:04.871525Z","shell.execute_reply":"2024-06-05T18:07:05.449600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[train_df['num_bbox'] > 0]\nprint(f\"Number of images with annotations: {len(train_df)}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:07.090866Z","iopub.execute_input":"2024-06-05T18:07:07.091346Z","iopub.status.idle":"2024-06-05T18:07:07.103466Z","shell.execute_reply.started":"2024-06-05T18:07:07.091294Z","shell.execute_reply":"2024-06-05T18:07:07.101735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['annotations'] = [ast.literal_eval(ann) for ann in train_df['annotations']]","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:09.543785Z","iopub.execute_input":"2024-06-05T18:07:09.544239Z","iopub.status.idle":"2024-06-05T18:07:09.865480Z","shell.execute_reply.started":"2024-06-05T18:07:09.544202Z","shell.execute_reply":"2024-06-05T18:07:09.863936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\nsplit = np.random.rand(len(train_df)) < 0.8\ntrain_df['split'] = np.where(split, 'train', 'valid')\n\ntrain_data = train_df[train_df['split'] == 'train'].reset_index(drop=True)\nvalid_data = train_df[train_df['split'] == 'valid'].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:12.378378Z","iopub.execute_input":"2024-06-05T18:07:12.378941Z","iopub.status.idle":"2024-06-05T18:07:12.399118Z","shell.execute_reply.started":"2024-06-05T18:07:12.378901Z","shell.execute_reply":"2024-06-05T18:07:12.397356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_count = len(train_data)\nvalid_count = len(valid_data)\nprint(f\"train data: {train_count}, valid data: {valid_count}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:14.548996Z","iopub.execute_input":"2024-06-05T18:07:14.549495Z","iopub.status.idle":"2024-06-05T18:07:14.556265Z","shell.execute_reply.started":"2024-06-05T18:07:14.549443Z","shell.execute_reply":"2024-06-05T18:07:14.554842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_DIR = 'barrier_reef/images'\nLABEL_DIR = 'barrier_reef/labels'\nos.makedirs(IMAGE_DIR + '/train', exist_ok=True)\nos.makedirs(IMAGE_DIR + '/valid', exist_ok=True)\nos.makedirs(LABEL_DIR + '/train', exist_ok=True)\nos.makedirs(LABEL_DIR + '/valid', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:07:17.529730Z","iopub.execute_input":"2024-06-05T18:07:17.530191Z","iopub.status.idle":"2024-06-05T18:07:17.538827Z","shell.execute_reply.started":"2024-06-05T18:07:17.530157Z","shell.execute_reply":"2024-06-05T18:07:17.537423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\ndef save_images_and_labels(df, split):\n    for idx, row in df.iterrows():\n        img_dst = os.path.join(IMAGE_DIR, split, f\"{row['image_id']}.jpg\")\n        lbl_dst = os.path.join(LABEL_DIR, split, f\"{row['image_id']}.txt\")\n        shutil.copy(row['path'], img_dst)\n        with open(lbl_dst, 'w') as f:\n            for ann in row['annotations']:\n                x_center = (ann['x'] + ann['width'] / 2) / IMG_SIZE\n                y_center = (ann['y'] + ann['height'] / 2) / IMG_SIZE\n                width = ann['width'] / IMG_SIZE\n                height = ann['height'] / IMG_SIZE\n                f.write(f\"0 {x_center} {y_center} {width} {height}\\n\")\n\n#save images\nsave_images_and_labels(train_data, 'train')\nsave_images_and_labels(valid_data, 'valid')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating yaml dir\nos.makedirs('tmp/yolov5/data', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:08:03.740523Z","iopub.execute_input":"2024-06-05T18:08:03.741024Z","iopub.status.idle":"2024-06-05T18:08:03.748166Z","shell.execute_reply.started":"2024-06-05T18:08:03.740982Z","shell.execute_reply":"2024-06-05T18:08:03.746660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import yaml  \n\ndata_yaml = {\n    'train': '../barrier_reef/images/train',\n    'val': '../barrier_reef/images/valid',\n    'nc': 1,\n    'names': ['cots']\n}\n\nwith open('tmp/yolov5/data/data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:08:05.422147Z","iopub.execute_input":"2024-06-05T18:08:05.422635Z","iopub.status.idle":"2024-06-05T18:08:05.431669Z","shell.execute_reply.started":"2024-06-05T18:08:05.422596Z","shell.execute_reply":"2024-06-05T18:08:05.430405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat tmp/yolov5/data/data.yaml","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:08:08.354665Z","iopub.execute_input":"2024-06-05T18:08:08.355171Z","iopub.status.idle":"2024-06-05T18:08:09.467497Z","shell.execute_reply.started":"2024-06-05T18:08:08.355132Z","shell.execute_reply":"2024-06-05T18:08:09.465813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yaml_file = 'tmp/yolov5/data/data.yaml'\n\n!python train.py --img 640 --batch 16 --epochs 10 --data {yaml_file} --weights yolov5s.pt\n","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:08:15.033556Z","iopub.execute_input":"2024-06-05T18:08:15.034939Z","iopub.status.idle":"2024-06-05T18:24:57.525467Z","shell.execute_reply.started":"2024-06-05T18:08:15.034884Z","shell.execute_reply":"2024-06-05T18:24:57.523333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python val.py --data {yaml_file} --weights runs/train/exp/weights/best.pt","metadata":{"execution":{"iopub.status.busy":"2024-06-05T18:25:01.830858Z","iopub.execute_input":"2024-06-05T18:25:01.831375Z","iopub.status.idle":"2024-06-05T18:25:50.246759Z","shell.execute_reply.started":"2024-06-05T18:25:01.831330Z","shell.execute_reply":"2024-06-05T18:25:50.245155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_path = 'runs/train/exp/results.csv'\nresults_df = pd.read_csv(results_path)\nresults_df.columns = results_df.columns.str.strip()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T19:45:35.205600Z","iopub.execute_input":"2024-06-05T19:45:35.206118Z","iopub.status.idle":"2024-06-05T19:45:35.218213Z","shell.execute_reply.started":"2024-06-05T19:45:35.206079Z","shell.execute_reply":"2024-06-05T19:45:35.216930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(results_df['epoch'], results_df['train/box_loss'], label='Box Loss')\nplt.plot(results_df['epoch'], results_df['train/obj_loss'], label='Object Loss')\nplt.plot(results_df['epoch'], results_df['train/cls_loss'], label='Class Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Training Losses')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T19:45:48.296851Z","iopub.execute_input":"2024-06-05T19:45:48.297359Z","iopub.status.idle":"2024-06-05T19:45:48.657414Z","shell.execute_reply.started":"2024-06-05T19:45:48.297300Z","shell.execute_reply":"2024-06-05T19:45:48.656099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(results_df['epoch'], results_df['metrics/precision'], label='Precision')\nplt.plot(results_df['epoch'], results_df['metrics/recall'], label='Recall')\nplt.xlabel('Epoch')\nplt.ylabel('Metric')\nplt.legend()\nplt.title('Precision and Recall')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T19:46:01.766796Z","iopub.execute_input":"2024-06-05T19:46:01.767208Z","iopub.status.idle":"2024-06-05T19:46:02.134843Z","shell.execute_reply.started":"2024-06-05T19:46:01.767177Z","shell.execute_reply":"2024-06-05T19:46:02.133415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(results_df['epoch'], results_df['metrics/mAP_0.5'], label='mAP@0.5')\nplt.plot(results_df['epoch'], results_df['metrics/mAP_0.5:0.95'], label='mAP@0.5:0.95')\nplt.xlabel('Epoch')\nplt.ylabel('mAP')\nplt.legend()\nplt.title('Mean Average Precision')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T19:46:16.900540Z","iopub.execute_input":"2024-06-05T19:46:16.900959Z","iopub.status.idle":"2024-06-05T19:46:17.194987Z","shell.execute_reply.started":"2024-06-05T19:46:16.900925Z","shell.execute_reply":"2024-06-05T19:46:17.193493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results_dir = 'runs/val/exp'\n\n# Get a list of image files in the results directory\nimage_files = glob.glob(os.path.join(results_dir, '*.jpg'))\n\n# Number of images to display in the collage\nnum_images = 6  # Adjust the number of images to display\ncols = 2  # Number of columns in the collage\nrows = (num_images + cols - 1) // cols  # Calculate the number of rows\n\n# Create a figure for the collage\nplt.figure(figsize=(15, 15))\n\n# Loop through the images and add them to the collage\nfor i, img_path in enumerate(image_files[:num_images]):\n    img = Image.open(img_path)\n    plt.subplot(rows, cols, i + 1)\n    plt.imshow(img)\n    plt.axis('off')\n\n# Display the collage\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T20:06:04.620637Z","iopub.execute_input":"2024-06-05T20:06:04.621112Z","iopub.status.idle":"2024-06-05T20:06:05.736402Z","shell.execute_reply.started":"2024-06-05T20:06:04.621068Z","shell.execute_reply":"2024-06-05T20:06:05.734389Z"},"trusted":true},"execution_count":null,"outputs":[]}]}