{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### CHECK PYTORCH VERSION","metadata":{}},{"cell_type":"code","source":"import torch\nprint(\"PyTorch version: \", torch.__version__)\nprint(\"GPU: \", torch.cuda.is_available())\nprint(\"Type: \", torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else \"CPU\")","metadata":{"_uuid":"6ce82614-2b05-4321-a92c-de45d1bad1f8","_cell_guid":"ec4fb542-14b9-4c41-982b-dd6432d69cfd","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-01-06T16:07:09.801917Z","iopub.execute_input":"2022-01-06T16:07:09.802728Z","iopub.status.idle":"2022-01-06T16:07:11.3055Z","shell.execute_reply.started":"2022-01-06T16:07:09.802684Z","shell.execute_reply":"2022-01-06T16:07:11.304611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### IMPORT LIBRARY","metadata":{}},{"cell_type":"code","source":"import ast\nimport glob\nimport os\nimport yaml\n\nimport numpy as np\nimport pandas as pd\n\n\nfrom IPython.display import Image, display\nfrom IPython.core.magic import register_line_cell_magic\nfrom shutil import copyfile\nfrom tqdm import tqdm\ntqdm.pandas()\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"c65091b9-a46f-4315-8994-55b213b155c8","_cell_guid":"d2c8398d-b169-46d6-9f67-6713fb8b114a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-01-06T16:07:11.307342Z","iopub.execute_input":"2022-01-06T16:07:11.307537Z","iopub.status.idle":"2022-01-06T16:07:11.334725Z","shell.execute_reply.started":"2022-01-06T16:07:11.307512Z","shell.execute_reply":"2022-01-06T16:07:11.334058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOME_DIR = '/kaggle/working'\nDATASET_PATH = '/kaggle/input/tensorflow-great-barrier-reef'","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:11.335788Z","iopub.execute_input":"2022-01-06T16:07:11.336054Z","iopub.status.idle":"2022-01-06T16:07:11.341322Z","shell.execute_reply.started":"2022-01-06T16:07:11.336019Z","shell.execute_reply":"2022-01-06T16:07:11.340602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. PREPARE DATASET","metadata":{}},{"cell_type":"code","source":"# I just used spllited dataset by @julian3833 - Reef - A CV strategy: subsequences! \n# https://www.kaggle.com/julian3833/reef-a-cv-strategy-subsequences \n\ndf = pd.read_csv(\"../input/reef-cv-strategy-subsequences-dataframes/train-validation-split/train-0.1.csv\")\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:11.342861Z","iopub.execute_input":"2022-01-06T16:07:11.343098Z","iopub.status.idle":"2022-01-06T16:07:11.464583Z","shell.execute_reply.started":"2022-01-06T16:07:11.343074Z","shell.execute_reply":"2022-01-06T16:07:11.46377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_path(row):\n    return f'{DATASET_PATH}/train_images/video_{row.video_id}/{row.video_frame}.jpg'\n\ndef num_boxes(annotations):\n    annotations = ast.literal_eval(annotations)\n    return len(annotations)\n\ndf['path'] = df.apply(lambda row: add_path(row), axis=1)\ndf['num_bbox'] = df['annotations'].apply(lambda x: num_boxes(x))\nprint(\"New path and annotations preprocessing completed\")\n\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:11.466001Z","iopub.execute_input":"2022-01-06T16:07:11.466247Z","iopub.status.idle":"2022-01-06T16:07:12.38385Z","shell.execute_reply.started":"2022-01-06T16:07:11.466214Z","shell.execute_reply":"2022-01-06T16:07:12.383159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df.num_bbox > 0]\n\nprint(f'Dataset images with annotations: {len(df)}')\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:25.985202Z","iopub.execute_input":"2022-01-06T16:07:25.985481Z","iopub.status.idle":"2022-01-06T16:07:26.007628Z","shell.execute_reply.started":"2022-01-06T16:07:25.985453Z","shell.execute_reply":"2022-01-06T16:07:26.006964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_new_path(row):\n    if row.is_train:\n        return f\"{HOME_DIR}/yolo_dataset/images/train/{row.image_id}.jpg\"\n    else:\n        return f\"{HOME_DIR}/yolo_dataset/images/valid/{row.image_id}.jpg\"\n    \ndf['new_path'] = df.apply(lambda row: add_new_path(row), axis=1)\nprint(\"New image path for train/valid created\")\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:28.906998Z","iopub.execute_input":"2022-01-06T16:07:28.907509Z","iopub.status.idle":"2022-01-06T16:07:29.045374Z","shell.execute_reply.started":"2022-01-06T16:07:28.907469Z","shell.execute_reply":"2022-01-06T16:07:29.044614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['path'][16])\nprint(df['new_path'][16])\nprint(df['image_path'][16])","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:29.046803Z","iopub.execute_input":"2022-01-06T16:07:29.04713Z","iopub.status.idle":"2022-01-06T16:07:29.055482Z","shell.execute_reply.started":"2022-01-06T16:07:29.047097Z","shell.execute_reply":"2022-01-06T16:07:29.054798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. CREATE DATASET FILE STRUCTURE","metadata":{}},{"cell_type":"code","source":"os.makedirs(f\"{HOME_DIR}/yolo_dataset/images/train\")\nos.makedirs(f\"{HOME_DIR}/yolo_dataset/images/valid\")\nos.makedirs(f\"{HOME_DIR}/yolo_dataset/labels/train\")\nos.makedirs(f\"{HOME_DIR}/yolo_dataset/labels/valid\")\nprint(f\"Directory structure yor Yolov5 created\")","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:29.056697Z","iopub.execute_input":"2022-01-06T16:07:29.057103Z","iopub.status.idle":"2022-01-06T16:07:29.065541Z","shell.execute_reply.started":"2022-01-06T16:07:29.057069Z","shell.execute_reply":"2022-01-06T16:07:29.064865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def copy_file(row):\n    copyfile(row.path, row.new_path)\n    \n_ = df.progress_apply(lambda row: copy_file(row), axis=1)\n\nprint(\"Sucessfully copy file for train and valid\")","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:07:29.067165Z","iopub.execute_input":"2022-01-06T16:07:29.067634Z","iopub.status.idle":"2022-01-06T16:08:17.166302Z","shell.execute_reply.started":"2022-01-06T16:07:29.067599Z","shell.execute_reply":"2022-01-06T16:08:17.165593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. CREATE YOLOv5 ANNOTATIONS","metadata":{}},{"cell_type":"code","source":"IMG_WIDTH, IMG_HEIGHT = 1280, 720\n\ndef get_yolo_format_bbox(bbox, img_w, img_h):\n    w = bbox['width']\n    h = bbox['height']\n    \n    if (bbox['x'] + bbox['width'] > img_w):\n        w = img_w - bbox['x']\n    if (bbox['y'] + bbox['height'] > img_h):\n        h = img_h - bbox['y']\n    \n    xc = bbox['x'] + int(np.round(w/2))\n    yc = bbox['y'] + int(np.round(h/2))\n    \n    # normalize\n    return [xc/img_w, yc/img_h, w/img_w, h/img_h]\n\nfor index, row in tqdm(df.iterrows()):\n    annotations = ast.literal_eval(row.annotations)\n    bboxes = []\n    for annot in annotations:\n        bbox = get_yolo_format_bbox(annot, IMG_WIDTH, IMG_HEIGHT)\n        bboxes.append(bbox)\n        \n    if row.is_train:\n        file_name = f\"{HOME_DIR}/yolo_dataset/labels/train/{row.image_id}.txt\"\n        os.makedirs(os.path.dirname(file_name), exist_ok=True)\n    else:\n        file_name = f\"{HOME_DIR}/yolo_dataset/labels/valid/{row.image_id}.txt\"\n        os.makedirs(os.path.dirname(file_name), exist_ok=True)\n        \n    with open(file_name, 'w') as f:\n        for i, bbox in enumerate(bboxes):\n            label = 0\n            bbox = [label] + bbox\n            bbox = [str(i) for i in bbox]\n            bbox = \" \".join(bbox)\n            f.write(bbox)\n            f.write(\"\\n\")\n\nprint(\"Annotations in Yolov5 format for all images created.\")","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:08:17.168119Z","iopub.execute_input":"2022-01-06T16:08:17.168596Z","iopub.status.idle":"2022-01-06T16:08:18.788948Z","shell.execute_reply.started":"2022-01-06T16:08:17.168544Z","shell.execute_reply":"2022-01-06T16:08:18.78525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. CREATE YOLOv5 DATASET CONFIGURATION FILE","metadata":{}},{"cell_type":"code","source":"train_data = os.listdir('/kaggle/working/yolo_dataset/labels/train')\nnum_train_file = len(train_data)\nprint(\"Number of txt file in train folder: \", num_train_file)\n\nvalid_data = os.listdir('/kaggle/working/yolo_dataset/labels/valid')\nnum_valid_file = len(valid_data)\nprint(\"Number of txt file in valid foler: \", num_valid_file)","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:08:18.790699Z","iopub.execute_input":"2022-01-06T16:08:18.791394Z","iopub.status.idle":"2022-01-06T16:08:18.800657Z","shell.execute_reply.started":"2022-01-06T16:08:18.791352Z","shell.execute_reply":"2022-01-06T16:08:18.799776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cat '/kaggle/working/yolo_dataset/labels/train/{train_data[10]}'","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:08:18.802409Z","iopub.execute_input":"2022-01-06T16:08:18.802723Z","iopub.status.idle":"2022-01-06T16:08:19.480161Z","shell.execute_reply.started":"2022-01-06T16:08:18.802686Z","shell.execute_reply":"2022-01-06T16:08:19.479285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. INSTALL YOLOv5\n\n4A. CLONE YOLOv5 GIT REPOSITORY","metadata":{}},{"cell_type":"code","source":"# Download YOLOv5\n# !git clone https://github.com/ultralytics/yolov5\n# !cp -r ../input/yolov5 ./\n!cp -r /kaggle/input/yolov5 /kaggle/working/\n!ls","metadata":{"execution":{"iopub.status.busy":"2022-01-06T16:21:31.571695Z","iopub.execute_input":"2022-01-06T16:21:31.572041Z","iopub.status.idle":"2022-01-06T16:21:32.985145Z","shell.execute_reply.started":"2022-01-06T16:21:31.571957Z","shell.execute_reply":"2022-01-06T16:21:32.984344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torchvision --upgrade -q\n!pip install wandb --upgrade","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd yolov5\n\n# Install dependencies\n!pip install -qr requirements.txt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"4B. CREATE YOLOv5 DATASET CONFIGURATION FILE","metadata":{}},{"cell_type":"code","source":"data_yaml = dict(\n    train = f\"{HOME_DIR}/yolo_dataset/images/train\",\n    val = f\"{HOME_DIR}/yolo_dataset/images/valid\",\n    nc = 1, # number of class\n    names = ['cots'] # classes\n)\n\nwith open(f'{HOME_DIR}/yolov5/data/data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=False)\n\nprint(\"Dataset configuration file for YOLOv5 is created\")\n\n%cat /kaggle/working/yolov5/data/data.yaml\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls '/kaggle/working/yolov5/data'","metadata":{"execution":{"iopub.status.busy":"2022-01-04T15:42:05.759854Z","iopub.execute_input":"2022-01-04T15:42:05.760539Z","iopub.status.idle":"2022-01-04T15:42:06.550502Z","shell.execute_reply.started":"2022-01-04T15:42:05.7605Z","shell.execute_reply":"2022-01-04T15:42:06.549493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# change directory\n%cd ..","metadata":{"execution":{"iopub.status.busy":"2022-01-04T15:42:09.369399Z","iopub.execute_input":"2022-01-04T15:42:09.369726Z","iopub.status.idle":"2022-01-04T15:42:09.378989Z","shell.execute_reply.started":"2022-01-04T15:42:09.369691Z","shell.execute_reply":"2022-01-04T15:42:09.377639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"4C. Train YOLOv5 with W&B","metadata":{}},{"cell_type":"code","source":"# more about Secrets -> https://www.kaggle.com/product-feedback/114053\nimport wandb\nfrom kaggle_secrets import UserSecretsClient\n\nuser_secrets = UserSecretsClient()\nwandb_api = user_secrets.get_secret(\"wandb_api\") \nwandb.login(key=wandb_api)\nwandb.login(anonymous='must')","metadata":{"execution":{"iopub.status.busy":"2022-01-04T15:42:26.699217Z","iopub.execute_input":"2022-01-04T15:42:26.699511Z","iopub.status.idle":"2022-01-04T15:42:29.007824Z","shell.execute_reply.started":"2022-01-04T15:42:26.699465Z","shell.execute_reply":"2022-01-04T15:42:29.006753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd yolov5","metadata":{"execution":{"iopub.status.busy":"2022-01-04T15:44:01.542297Z","iopub.execute_input":"2022-01-04T15:44:01.542611Z","iopub.status.idle":"2022-01-04T15:44:01.550878Z","shell.execute_reply.started":"2022-01-04T15:44:01.542565Z","shell.execute_reply":"2022-01-04T15:44:01.549477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 4\nEPOCHS = 20\nIMG_SIZE=1280\n# Selected_Fold=4  #0..4","metadata":{"execution":{"iopub.status.busy":"2022-01-04T15:44:12.858715Z","iopub.execute_input":"2022-01-04T15:44:12.859024Z","iopub.status.idle":"2022-01-04T15:44:12.863989Z","shell.execute_reply.started":"2022-01-04T15:44:12.858988Z","shell.execute_reply":"2022-01-04T15:44:12.862911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All training results are saved to runs/train/ with incrementing run directories, i.e. runs/train/exp2, runs/train/exp3 etc.","metadata":{}},{"cell_type":"code","source":"#best_weights = '/kaggle/input/nfl-weights/yolov5/kaggle-reef/exp/weights/best.pt' --weights {best_weights} \\\n!python train.py --img {IMG_SIZE} \\\n                 --batch {BATCH_SIZE} \\\n                 --epochs {EPOCHS} \\\n                 --data data.yaml \\\n                 --weights yolov5l6.pt \\\n                 --project kaggle-Reef \\\n                 --device 0 \\\n#                  --evolve","metadata":{},"execution_count":null,"outputs":[]}]}