{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\nThis notebook shows some characteristics of the images and labels used for the two contests. I've taken a small subset of training images and files from the Open Images dataset and put them here: Excerpt from OpenImages 2020 Train.\n\n### Some specific objectives:\n\n* Get a feel for the the images and the objects/segments they contain.\n* Implement some basic object detection.\n* Look at label counts image sizes, and object relationships.\n\nOn the technical side, there are some things you might find useful:\n\n* Modeling with large datasets in Kaggle notebooks\n* Making interactive plots with hvplot\n* Visualizing graph networks with networkX","metadata":{}},{"cell_type":"code","source":"!pip install pandas==1.3.0\n!pip install --upgrade scikit-learn\n!pip install typing-extensions --upgrade\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nfrom pathlib import Path\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n# import hvplot.pandas","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-03T12:05:24.528421Z","iopub.execute_input":"2022-12-03T12:05:24.530631Z","iopub.status.idle":"2022-12-03T12:05:24.564016Z","shell.execute_reply.started":"2022-12-03T12:05:24.530576Z","shell.execute_reply":"2022-12-03T12:05:24.563216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Images and annotations\nAnnotations and classes are different for the two competitions. The detection dataset has more classes of objects and more objects per image most of the time. In other words, images common to both challenges will have more boxes than masks.\n\nHere are a few images containing boxes and segment masks along with labels. I limited it to 6 for display purposes. It's easy enough to browse through hundreds and get a more complete idea.","metadata":{}},{"cell_type":"code","source":"data_dir = Path('../input/excerpt-from-openimages-2020-train')\nim_list = sorted(data_dir.glob('train_00_part/*.jpg'))\nmask_list = sorted(data_dir.glob('train-masks-f/*.png'))\nboxes_df = pd.read_csv(data_dir/'oidv6-train-annotations-bbox.csv')\n\nnames_ = ['LabelName', 'Label']\nlabels =  pd.read_csv(data_dir/'class-descriptions-boxable.csv', names=names_)\n\nim_ids = [im.stem for im in im_list]\ncols = ['ImageID', 'LabelName', 'XMin', 'YMin', 'XMax', 'YMax']\nboxes_df = boxes_df.loc[boxes_df.ImageID.isin(im_ids), cols] \\\n                   .merge(labels, how='left', on='LabelName')\nboxes_df","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-12-03T12:05:24.575199Z","iopub.execute_input":"2022-12-03T12:05:24.577435Z","iopub.status.idle":"2022-12-03T12:06:49.587851Z","shell.execute_reply.started":"2022-12-03T12:05:24.577395Z","shell.execute_reply":"2022-12-03T12:06:49.586876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n\nle = preprocessing.LabelEncoder()\nle.fit(boxes_df.Label)\nboxes_df['Label_Id'] = le.transform(boxes_df.Label)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:06:49.59022Z","iopub.execute_input":"2022-12-03T12:06:49.590921Z","iopub.status.idle":"2022-12-03T12:06:49.79197Z","shell.execute_reply.started":"2022-12-03T12:06:49.590877Z","shell.execute_reply":"2022-12-03T12:06:49.791078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Annotate and plot\n# cols, rows  = 3, 2\n# plt.figure(figsize=(20,30))\nall_images = list(boxes_df['ImageID'].unique())\nimage_ids = []\nclass_ids = [] \nclass_names = []\nwidths = []\nheights = [] \nw_boxs = []\nh_boxs = [] \nx_mins = [] \ny_mins = [] \nx_maxs = []\ny_maxs = [] \n\n\nfor image_name in all_images:\n    tmp = boxes_df[boxes_df['ImageID']==image_name].reset_index(drop=True)\n    \n    # info image \n    img = cv2.imread(os.path.join(\"../input/excerpt-from-openimages-2020-train/train_00_part/\",str(image_name)+\".jpg\"))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    h0, w0 = img.shape[:2]\n    \n    for idx in range(len(tmp)):\n        row = tmp.iloc[idx]\n        x_min = int(row[\"XMin\"] * w0)\n        x_max = int(row[\"XMax\"] * w0)\n        y_min = int(row[\"YMin\"] * h0)\n        y_max = int(row[\"YMax\"] * h0)\n        label_name = str(row['Label'])\n        label_id = int(row['Label_Id'])\n        \n        image_ids.append(image_name)\n        class_ids.append(label_id)\n        class_names.append(label_name)\n        widths.append(w0)\n        heights.append(h0)\n        \n        w_boxs.append(x_max-x_min)\n        h_boxs.append(y_max-y_min)\n        x_mins.append(x_min)\n        y_mins.append(y_min)\n        x_maxs.append(x_max)\n        y_maxs.append(y_max)\n\ntrain_df = {\"image_id\": image_ids, \"class_id\":class_ids, \n             \"class_name\": class_names, \"x_min\": x_mins, \n             \"x_max\": x_maxs, \"y_min\": y_mins, \"y_max\":y_maxs, \"w_box\": w_boxs, \"h_box\": h_boxs, \n            \"width\": widths, \"height\":heights}\ntrain_df = pd.DataFrame(train_df)\n\n\ndef get_images_path(image_id):\n    return os.path.join(\"../input/excerpt-from-openimages-2020-train/train_00_part/\", image_id+\".jpg\")\ntrain_df['image_path'] = train_df['image_id'].apply(get_images_path)\n\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:06:49.793844Z","iopub.execute_input":"2022-12-03T12:06:49.794206Z","iopub.status.idle":"2022-12-03T12:06:53.809118Z","shell.execute_reply.started":"2022-12-03T12:06:49.794167Z","shell.execute_reply":"2022-12-03T12:06:53.808133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['x_mid'] = train_df.apply(lambda row: (row.x_min/row.width + (row.x_min/row.width + row.w_box/row.width))/2 , axis =1)\ntrain_df['y_mid'] = train_df.apply(lambda row: (row.y_min/row.height + (row.y_min/row.height+ row.h_box/row.height))/2, axis =1)\ntrain_df['w'] = train_df.apply(lambda row: row.w_box/row.width, axis =1)\ntrain_df['h'] = train_df.apply(lambda row: row.h_box/row.height, axis =1)\ntrain_df['area'] = train_df['w']*train_df['h']","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:06:53.810496Z","iopub.execute_input":"2022-12-03T12:06:53.81088Z","iopub.status.idle":"2022-12-03T12:06:54.170481Z","shell.execute_reply.started":"2022-12-03T12:06:53.810841Z","shell.execute_reply":"2022-12-03T12:06:54.169313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv(\"./train_detection.csv\", index=False)\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:06:54.172005Z","iopub.execute_input":"2022-12-03T12:06:54.172538Z","iopub.status.idle":"2022-12-03T12:06:54.222182Z","shell.execute_reply.started":"2022-12-03T12:06:54.172494Z","shell.execute_reply":"2022-12-03T12:06:54.220971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedGroupKFold \nfrom os import listdir\nfrom os.path import isfile, join\nimport yaml\n\n\nfold_num = 1 ## -1 is all train and all val \nn_folds  = 2\nSEED = 46\n\n# train_smoke_wildlife = pd.read_csv(join('../data/wildfire-smoke-detection-camera/input', 'train_detection.csv')).reset_index(drop=True)\n# train_smoke_large    = pd.read_csv(\"../data/SmokeDataset/smoke100k-L/train_detection.csv\").reset_index(drop=True)\n\n# # train_fire_csv  = pd.read_csv(join('../data/Base/fire_detection/VOC2020', 'train_detection.csv')).reset_index(drop=True)\n# train_df = pd.concat([train_smoke_wildlife, train_smoke_large]).reset_index(drop=True)\n# train_df = pd.read_csv(join('../data/wildfire-smoke-detection-camera/input', 'train_detection.csv')).reset_index(drop=True)\ntrain_df = pd.read_csv(\"./train_detection.csv\").reset_index(drop=False)\nRoot_dir    = f'./LARGE-YOLOV5x6-Fold{fold_num}'\n\nlabel_dir   = join(Root_dir,\"All_Labels\")\nlabel_train = join(Root_dir,'labels/train') \nlabel_val   = join(Root_dir,'labels/val')\nimage_train = join(Root_dir,'images/train')\nimage_val   = join(Root_dir,'images/val')\n\nos.makedirs(label_dir, exist_ok = True)\nos.makedirs(label_train, exist_ok = True)\nos.makedirs(label_val, exist_ok = True)\nos.makedirs(image_train, exist_ok = True)\nos.makedirs(image_val, exist_ok = True)\n\n\n# ============= Convert All Labels For All Images & Save To 1 Folder =============\nfor img_id in tqdm(list(train_df.image_id.unique())):\n    list_img_id = train_df[train_df[\"image_id\"] == img_id].reset_index(drop=True)\n    name_file = img_id + \".txt\"\n    with open(os.path.join(label_dir, name_file), \"w\") as f:\n        for idx in range(len(list_img_id)):\n            f.write(str(list_img_id.iloc[idx][\"class_id\"]))\n            f.write(\" \")\n            f.write('{}'.format(list_img_id.iloc[idx][\"x_mid\"]))\n            f.write(\" \")\n            f.write('{}'.format(list_img_id.iloc[idx][\"y_mid\"]))\n            f.write(\" \")\n            f.write('{}'.format(list_img_id.iloc[idx][\"w\"]))\n            f.write(\" \")\n            f.write('{}'.format(list_img_id.iloc[idx][\"h\"]))\n            f.write(\"\\n\")\n\n\n# ============= Fold Split =============\ngkf  = StratifiedGroupKFold(n_splits = n_folds, random_state=SEED, shuffle=True)\ntrain_df['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train_df, train_df.class_id.tolist(), groups = train_df.image_id.tolist())):\n    train_df.loc[val_idx, 'fold'] = fold\n\ntrain_files = []\nval_files   = []\n\n# if fold_num == -1:\n#     train_files += list(train_df.image_path.unique())\n#     val_files   += list(train_df.image_path.unique())\n# else:\n#     train_files += list(train_df[train_df.fold!=fold_num].image_path.unique())\n#     val_files += list(train_df[train_df.fold==fold_num].image_path.unique())\n\ntrain_files += list(train_df.image_path.unique())\nval_files   += list(train_df.image_path.unique())\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:07:27.476444Z","iopub.execute_input":"2022-12-03T12:07:27.476833Z","iopub.status.idle":"2022-12-03T12:07:30.207814Z","shell.execute_reply.started":"2022-12-03T12:07:27.476791Z","shell.execute_reply":"2022-12-03T12:07:30.206662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ============= Move File =============\nfor file in tqdm(train_files):\n    shutil.copy(file, image_train)\n    filename = file.split('/')[-1].split('.')[0]\n    shutil.copy(os.path.join(label_dir, filename+'.txt'), label_train)\n    \nfor file in tqdm(val_files):\n    shutil.copy(file, image_val)\n    filename = file.split('/')[-1].split('.')[0]\n    shutil.copy(os.path.join(label_dir, filename+'.txt'), label_val)\n\n    \n# ============== Class Id =============\nclass_ids, class_names = list(zip(*set(zip(train_df.class_id, train_df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\n\n\n# ============= Make train.txt + val.txt =============\ncwd = os.path.abspath(Root_dir)\nwith open(join(cwd , 'train.txt'), 'w') as f:\n    tmp_dir = os.path.abspath(join(os.getcwd(), image_train))\n    for path in os.listdir(tmp_dir):\n        f.write(join(tmp_dir, path)+'\\n')\n            \nwith open(join(cwd , 'val.txt'), 'w') as f:\n    tmp_dir = os.path.abspath(join(os.getcwd(), image_val))\n    for path in os.listdir(tmp_dir):\n        f.write(join(tmp_dir, path)+'\\n')\n\ndata = dict(\n    train =  join(cwd, 'train.txt') ,\n#     val   =  join(cwd, 'val.txt' ),\n    val   =  join(cwd, 'train.txt' ),\n    nc    = len(classes),\n    names = list(classes)\n)\n\nwith open(join(cwd,'LARGE-YOLOV5x6.yaml'), 'w') as outfile:\n    yaml.dump(data, outfile, default_flow_style=False)\n\nf = open(join(cwd,'LARGE-YOLOV5x6.yaml'), 'r')\nprint('\\nyaml:')\nprint(f.read())","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:15:44.051068Z","iopub.execute_input":"2022-12-03T12:15:44.051429Z","iopub.status.idle":"2022-12-03T12:15:44.945467Z","shell.execute_reply.started":"2022-12-03T12:15:44.051397Z","shell.execute_reply":"2022-12-03T12:15:44.944265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/ultralytics/yolov5.git\n!pip install -r yolov5/requirements.txt","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:07:52.003214Z","iopub.execute_input":"2022-12-03T12:07:52.003566Z","iopub.status.idle":"2022-12-03T12:08:00.60863Z","shell.execute_reply.started":"2022-12-03T12:07:52.00353Z","shell.execute_reply":"2022-12-03T12:08:00.607395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nos.chdir(\"yolov5\")","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:15:51.571839Z","iopub.execute_input":"2022-12-03T12:15:51.572201Z","iopub.status.idle":"2022-12-03T12:15:51.576307Z","shell.execute_reply.started":"2022-12-03T12:15:51.572164Z","shell.execute_reply":"2022-12-03T12:15:51.575409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import ImageFile\nImageFile.LOAD_TRUNCATED_IMAGES = True","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:15:51.719564Z","iopub.execute_input":"2022-12-03T12:15:51.719836Z","iopub.status.idle":"2022-12-03T12:15:51.725292Z","shell.execute_reply.started":"2022-12-03T12:15:51.719809Z","shell.execute_reply":"2022-12-03T12:15:51.724311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wandb off","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:20:48.130377Z","iopub.execute_input":"2022-12-03T12:20:48.130734Z","iopub.status.idle":"2022-12-03T12:20:50.111707Z","shell.execute_reply.started":"2022-12-03T12:20:48.130701Z","shell.execute_reply":"2022-12-03T12:20:50.110172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!python train.py --img 640\\\n--batch 4\\\n--epochs 50\\\n--data ../LARGE-YOLOV5x6-Fold1/LARGE-YOLOV5x6.yaml\\\n--workers 4\\\n--weights yolov5x6.pt","metadata":{"execution":{"iopub.status.busy":"2022-12-03T12:20:52.940044Z","iopub.execute_input":"2022-12-03T12:20:52.940412Z","iopub.status.idle":"2022-12-03T12:22:18.150192Z","shell.execute_reply.started":"2022-12-03T12:20:52.940375Z","shell.execute_reply":"2022-12-03T12:22:18.149059Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"../LARGE-YOLOV5x6-Fold0/labels/val/fd9e6a44764744e4.txt\", \"r\") as f:\n    hi = f.read()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check1 = os.listdir(\"../LARGE-YOLOV5x6-Fold0/labels/train/\")\ncheck2 = \n\nlen(check1)\nlen([item for item in check1 if item.replace(\".txt\", \".jpg\") not in check2])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for image_name in os.listdir(\"../LARGE-YOLOV5x6-Fold0/images/val\"):\n#     tmp = boxes_df[boxes_df['ImageID']==image_name].reset_index(drop=True)\n    \n    # info image \n    img = cv2.imread(os.path.join(\"../LARGE-YOLOV5x6-Fold0/images/val\",str(image_name)))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    print(img.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls ./runs/train/exp5/train_batch0.jpg","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = cv2.imread(\"./runs/train/exp5/train_batch0.jpg\")\n\nplt.imshow(img)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Format yolov5 format ","metadata":{}},{"cell_type":"markdown","source":"## ------------------Done the Notebook-------------------","metadata":{}}]}