{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-09T16:31:51.608185Z","iopub.execute_input":"2023-11-09T16:31:51.608754Z","iopub.status.idle":"2023-11-09T16:32:13.263354Z","shell.execute_reply.started":"2023-11-09T16:31:51.608705Z","shell.execute_reply":"2023-11-09T16:32:13.262443Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:13.265303Z","iopub.execute_input":"2023-11-09T16:32:13.266137Z","iopub.status.idle":"2023-11-09T16:32:13.270317Z","shell.execute_reply.started":"2023-11-09T16:32:13.266102Z","shell.execute_reply":"2023-11-09T16:32:13.269438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/tensorflow-great-barrier-reef/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:13.271436Z","iopub.execute_input":"2023-11-09T16:32:13.271744Z","iopub.status.idle":"2023-11-09T16:32:13.3276Z","shell.execute_reply.started":"2023-11-09T16:32:13.271715Z","shell.execute_reply":"2023-11-09T16:32:13.326804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:13.329444Z","iopub.execute_input":"2023-11-09T16:32:13.329727Z","iopub.status.idle":"2023-11-09T16:32:13.354322Z","shell.execute_reply.started":"2023-11-09T16:32:13.329703Z","shell.execute_reply":"2023-11-09T16:32:13.353463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/tensorflow-great-barrier-reef/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:13.355462Z","iopub.execute_input":"2023-11-09T16:32:13.355817Z","iopub.status.idle":"2023-11-09T16:32:13.363735Z","shell.execute_reply.started":"2023-11-09T16:32:13.355781Z","shell.execute_reply":"2023-11-09T16:32:13.362828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:13.364903Z","iopub.execute_input":"2023-11-09T16:32:13.365507Z","iopub.status.idle":"2023-11-09T16:32:13.374877Z","shell.execute_reply.started":"2023-11-09T16:32:13.365472Z","shell.execute_reply":"2023-11-09T16:32:13.373942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:13.376201Z","iopub.execute_input":"2023-11-09T16:32:13.376526Z","iopub.status.idle":"2023-11-09T16:32:27.128953Z","shell.execute_reply.started":"2023-11-09T16:32:13.376499Z","shell.execute_reply":"2023-11-09T16:32:27.127768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:27.130551Z","iopub.execute_input":"2023-11-09T16:32:27.130912Z","iopub.status.idle":"2023-11-09T16:32:31.144055Z","shell.execute_reply.started":"2023-11-09T16:32:27.130879Z","shell.execute_reply":"2023-11-09T16:32:31.143275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO(\"yolov8n.yaml\")","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.145162Z","iopub.execute_input":"2023-11-09T16:32:31.145577Z","iopub.status.idle":"2023-11-09T16:32:31.67285Z","shell.execute_reply.started":"2023-11-09T16:32:31.14555Z","shell.execute_reply":"2023-11-09T16:32:31.671967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['images_path'] = '/kaggle/input/tensorflow-great-barrier-reef/train_images/video_'+df['video_id'].astype(str)+'/'+df[\"image_id\"].apply(lambda x: x.split(\"-\")[1])+'.jpg'","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.676343Z","iopub.execute_input":"2023-11-09T16:32:31.676594Z","iopub.status.idle":"2023-11-09T16:32:31.724799Z","shell.execute_reply.started":"2023-11-09T16:32:31.676572Z","shell.execute_reply":"2023-11-09T16:32:31.724096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.725835Z","iopub.execute_input":"2023-11-09T16:32:31.726117Z","iopub.status.idle":"2023-11-09T16:32:31.740562Z","shell.execute_reply.started":"2023-11-09T16:32:31.726093Z","shell.execute_reply":"2023-11-09T16:32:31.739592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.742073Z","iopub.execute_input":"2023-11-09T16:32:31.742437Z","iopub.status.idle":"2023-11-09T16:32:31.762403Z","shell.execute_reply.started":"2023-11-09T16:32:31.742397Z","shell.execute_reply":"2023-11-09T16:32:31.761519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.annotations.unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.763405Z","iopub.execute_input":"2023-11-09T16:32:31.763662Z","iopub.status.idle":"2023-11-09T16:32:31.776131Z","shell.execute_reply.started":"2023-11-09T16:32:31.763639Z","shell.execute_reply":"2023-11-09T16:32:31.775272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\nimage = cv2.imread(df['images_path'][0]) \nheight, width, _ = image.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.777252Z","iopub.execute_input":"2023-11-09T16:32:31.777519Z","iopub.status.idle":"2023-11-09T16:32:31.829156Z","shell.execute_reply.started":"2023-11-09T16:32:31.777495Z","shell.execute_reply":"2023-11-09T16:32:31.828241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(height, width)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.830377Z","iopub.execute_input":"2023-11-09T16:32:31.830717Z","iopub.status.idle":"2023-11-09T16:32:31.83578Z","shell.execute_reply.started":"2023-11-09T16:32:31.830684Z","shell.execute_reply":"2023-11-09T16:32:31.834876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['annot'] = True\nfor i in range(len(df['annotations'])):\n    if len(df['annotations'][i])==2:\n        df.iloc[i, df.columns.get_loc('annot')] = False\n    else:\n        df.iloc[i, df.columns.get_loc('annot')] = True","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:31.836945Z","iopub.execute_input":"2023-11-09T16:32:31.837269Z","iopub.status.idle":"2023-11-09T16:32:34.267188Z","shell.execute_reply.started":"2023-11-09T16:32:31.837244Z","shell.execute_reply":"2023-11-09T16:32:34.266193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.268292Z","iopub.execute_input":"2023-11-09T16:32:34.268597Z","iopub.status.idle":"2023-11-09T16:32:34.284763Z","shell.execute_reply.started":"2023-11-09T16:32:34.268571Z","shell.execute_reply":"2023-11-09T16:32:34.283787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['video_frame'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.286119Z","iopub.execute_input":"2023-11-09T16:32:34.28647Z","iopub.status.idle":"2023-11-09T16:32:34.296777Z","shell.execute_reply.started":"2023-11-09T16:32:34.286435Z","shell.execute_reply":"2023-11-09T16:32:34.295779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def format_annot(x):\n    \n    annotations = eval(x)\n    new_annotations = ''\n    img_h = height\n    img_w = width\n    \n    if len(annotations) != 2:\n        for annot in annotations:\n            category = 0\n            x = annot['x']\n            y = annot['y']\n            w = annot['width']\n            h = annot['height']\n            \n            x_center = x + (w/2)\n            y_center = y + (h/2)\n            \n            #Normalization\n            x_center = x_center/img_w\n            y_center = y_center/img_h\n            \n            w = w/img_w\n            h = h/img_h\n            \n            x_centre = format(x_center, '.6f')\n            y_centre = format(y_center, '.6f')\n            w = format(w, '.6f')\n            h = format(h, '.6f')\n            \n            new_annotations = new_annotations + (f\"{category} {x_center} {y_center} {w} {h}\\n\")\n        \n        return new_annotations \n    else:\n        return '' ","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.298071Z","iopub.execute_input":"2023-11-09T16:32:34.298381Z","iopub.status.idle":"2023-11-09T16:32:34.307005Z","shell.execute_reply.started":"2023-11-09T16:32:34.298355Z","shell.execute_reply":"2023-11-09T16:32:34.305997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['yolo'] = df['annotations'].apply(lambda x: format_annot(x))","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.308144Z","iopub.execute_input":"2023-11-09T16:32:34.308393Z","iopub.status.idle":"2023-11-09T16:32:34.689882Z","shell.execute_reply.started":"2023-11-09T16:32:34.30837Z","shell.execute_reply":"2023-11-09T16:32:34.688913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.690934Z","iopub.execute_input":"2023-11-09T16:32:34.691247Z","iopub.status.idle":"2023-11-09T16:32:34.707358Z","shell.execute_reply.started":"2023-11-09T16:32:34.691216Z","shell.execute_reply":"2023-11-09T16:32:34.706317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_path = '/kaggle/working/train/images'\nif not os.path.exists(images_path):\n    os.makedirs(images_path)\n    \nlabel_path = '/kaggle/working/train/labels'\nif not os.path.exists(label_path):\n    os.makedirs(label_path)\n\ntest_images_path =  '/kaggle/working/test/images'   \nif not os.path.exists(test_images_path):\n    os.makedirs(test_images_path)  \n\ntest_labels_path =  '/kaggle/working/test/labels'   \nif not os.path.exists(test_labels_path):\n    os.makedirs(test_labels_path)    ","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.708513Z","iopub.execute_input":"2023-11-09T16:32:34.708808Z","iopub.status.idle":"2023-11-09T16:32:34.717213Z","shell.execute_reply.started":"2023-11-09T16:32:34.708784Z","shell.execute_reply":"2023-11-09T16:32:34.716139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n\nannoted_image_paths = []\nannoted_labels = []\n\nfor i in range(len(df['annotations'])):\n    if len(df['annotations'][i])!=2:\n        path = df.iloc[i, df.columns.get_loc('images_path')]\n        annoted_image_paths.append(path)\n        lab = df.iloc[i, df.columns.get_loc('yolo')]\n        annoted_labels.append(lab)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:34.718413Z","iopub.execute_input":"2023-11-09T16:32:34.718704Z","iopub.status.idle":"2023-11-09T16:32:35.264423Z","shell.execute_reply.started":"2023-11-09T16:32:34.71868Z","shell.execute_reply":"2023-11-09T16:32:35.263364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df = pd.DataFrame({'image_path':annoted_image_paths, 'labels':annoted_labels})","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:35.265646Z","iopub.execute_input":"2023-11-09T16:32:35.265913Z","iopub.status.idle":"2023-11-09T16:32:35.271775Z","shell.execute_reply.started":"2023-11-09T16:32:35.265891Z","shell.execute_reply":"2023-11-09T16:32:35.270762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df['labels'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:35.272888Z","iopub.execute_input":"2023-11-09T16:32:35.273199Z","iopub.status.idle":"2023-11-09T16:32:35.286351Z","shell.execute_reply.started":"2023-11-09T16:32:35.273176Z","shell.execute_reply":"2023-11-09T16:32:35.285428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_dir = '/kaggle/input/tensorflow-great-barrier-reef/train_images/'\nimg_gen_dir = '/kaggle/working/train/images/'\nfount = 0\n\nfor folder in os.listdir(main_dir):\n    folders_path = os.path.join(main_dir, folder)\n    for folders in os.listdir(folders_path):\n        file_path = os.path.join(folders_path, folders)\n        if file_path in annoted_image_paths:\n            fount += 1\n            shutil.copy(file_path, img_gen_dir)\nprint(fount)            ","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:32:35.287467Z","iopub.execute_input":"2023-11-09T16:32:35.287792Z","iopub.status.idle":"2023-11-09T16:33:36.863273Z","shell.execute_reply.started":"2023-11-09T16:32:35.287756Z","shell.execute_reply":"2023-11-09T16:33:36.862262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lab_gen_dir = '/kaggle/working/train/labels/'\ncount = 0\nfor i in range(len(new_df['labels'])):\n    p = new_df.iloc[i, new_df.columns.get_loc('image_path')]\n    _, filename = os.path.split(p)\n    name = filename.split('.')[0]\n    name = name + '.txt'\n    count += 1\n    file = open(os.path.join(lab_gen_dir, name), 'w')\n    file.write(new_df.iloc[i, new_df.columns.get_loc('labels')])\n    file.close\nprint(count)    ","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:33:36.864842Z","iopub.execute_input":"2023-11-09T16:33:36.86514Z","iopub.status.idle":"2023-11-09T16:33:37.581648Z","shell.execute_reply.started":"2023-11-09T16:33:36.865112Z","shell.execute_reply":"2023-11-09T16:33:37.580695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n\nimage_paths = glob.glob(img_gen_dir + '*.jpg')\nlabel_paths = glob.glob(lab_gen_dir + '*.txt')\n\nfrom sklearn.model_selection import train_test_split\n\ntrain_image_paths, test_images_paths, train_label_paths, test_label_paths = train_test_split(image_paths, label_paths, test_size = 0.2, random_state=42)\n\ndef move_files(source_paths, destination_dir):\n    for source_path in source_paths:\n        filename = os.path.basename(source_path)\n        destination_path = os.path.join(destination_dir, filename)\n        shutil.move(source_path, destination_path)\n\ntrain_image_destination = '/kaggle/working/train/images/'\ntrain_label_destination = '/kaggle/working/train/labels/'\ntest_image_destination = '/kaggle/working/test/images/'\ntest_label_destination = '/kaggle/working/test/labels/'\n\nmove_files(train_image_paths, train_image_destination) \nmove_files(train_label_paths, train_label_destination)\nmove_files(test_images_paths,test_image_destination)\nmove_files(test_label_paths, test_label_destination)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:33:37.585766Z","iopub.execute_input":"2023-11-09T16:33:37.586074Z","iopub.status.idle":"2023-11-09T16:33:38.252353Z","shell.execute_reply.started":"2023-11-09T16:33:37.586049Z","shell.execute_reply":"2023-11-09T16:33:38.251519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import yaml\n\nconfig = {\n    \"train\": \"/kaggle/working/train/images/\",\n    \"val\": \"/kaggle/working/test/images/\",\n    \n    \"nc\": 1 \n    \n}\nyml_file_path = '/kaggle/working/config.yml'","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:33:38.253446Z","iopub.execute_input":"2023-11-09T16:33:38.253734Z","iopub.status.idle":"2023-11-09T16:33:38.258384Z","shell.execute_reply.started":"2023-11-09T16:33:38.25371Z","shell.execute_reply":"2023-11-09T16:33:38.257396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(yml_file_path, 'w') as yml_file:\n    yaml.dump(config, yml_file, default_flow_style=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:33:38.259577Z","iopub.execute_input":"2023-11-09T16:33:38.25993Z","iopub.status.idle":"2023-11-09T16:33:38.27223Z","shell.execute_reply.started":"2023-11-09T16:33:38.259896Z","shell.execute_reply":"2023-11-09T16:33:38.271472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.train(data = \"/kaggle/working/config.yml\", epochs=100)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T16:33:38.273421Z","iopub.execute_input":"2023-11-09T16:33:38.273764Z","iopub.status.idle":"2023-11-09T18:39:52.653493Z","shell.execute_reply.started":"2023-11-09T16:33:38.27373Z","shell.execute_reply":"2023-11-09T18:39:52.652071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}