{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### **import dependencies**","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nimport yaml\n\nfrom kaggle_secrets import UserSecretsClient\nimport wandb\nfrom wandb.keras import WandbCallback\n\nimport cv2\nimport pydicom\n\nfrom pathlib import Path\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-06-17T08:52:50.222412Z","iopub.execute_input":"2021-06-17T08:52:50.222875Z","iopub.status.idle":"2021-06-17T08:52:58.476892Z","shell.execute_reply.started":"2021-06-17T08:52:50.222778Z","shell.execute_reply":"2021-06-17T08:52:58.475791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **configuration and initialization**","metadata":{}},{"cell_type":"code","source":"SIIM_COVID19_DETECTION_DIR = '/kaggle/input/siim-covid19-detection/'\nPART0_RESIZED_DIR = '/kaggle/input/part0-siim-covid19-first-look-resized-512px/'\nYOLOV5_DIR = '/kaggle/input/yolov5/yolov5/'\nYOLOV5_W_DIR = '/kaggle/working/yolov5/yolov5/'\n\nTEMP_DIR = '/kaggle/temp/'\nINPUT_DIR = PART0_RESIZED_DIR+'data/'\nOUTPUT_DIR = DATASET_DIR = TEMP_DIR+'data/'\n\nTRAIN_IMAGES_DIR = DATASET_DIR + 'images/train/'\nVAL_IMAGES_DIR = DATASET_DIR +'images/valid/'\nTRAIN_LABELS_DIR = DATASET_DIR + 'labels/train/'\nVAL_LABELS_DIR = DATASET_DIR +'labels/valid/'\n\nBATCH_SIZE = 8\nEPOCHS = 50\nIMG_SIZE = WIDTH = HEIGHT = 512\n\nTRAIN_IMAGE_LEVEL_PATH = SIIM_COVID19_DETECTION_DIR+'train_image_level.csv'\nTRAIN_STUDY_LEVEL_PATH = SIIM_COVID19_DETECTION_DIR+'train_study_level.csv'\nMETA_PATH = PART0_RESIZED_DIR+'meta.csv'\n\nINTERPOLATION = cv2.INTER_LANCZOS4\n\nWANDB_PROJECT_NAME = 'project8-kaggle-covid19'\nWANDB_ENTITY_NAME = ''","metadata":{"execution":{"iopub.status.busy":"2021-06-17T08:52:58.478489Z","iopub.execute_input":"2021-06-17T08:52:58.478755Z","iopub.status.idle":"2021-06-17T08:52:58.485496Z","shell.execute_reply.started":"2021-06-17T08:52:58.478726Z","shell.execute_reply":"2021-06-17T08:52:58.484453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"WANDB_API_KEY1\")\nos.environ['WANDB_API_KEY'] = secret_value_0\n\n#wandb.login()\nwandb.init(project=WANDB_PROJECT_NAME)\nconfig = wandb.config \nconfig.batch_size = BATCH_SIZE\n\n%cd ../../\n\nos.makedirs(TRAIN_IMAGES_DIR, exist_ok=True)\nos.makedirs(VAL_IMAGES_DIR, exist_ok=True)\nos.makedirs(TRAIN_LABELS_DIR, exist_ok=True)\nos.makedirs(VAL_LABELS_DIR, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-17T08:53:04.177246Z","iopub.execute_input":"2021-06-17T08:53:04.177862Z","iopub.status.idle":"2021-06-17T08:53:09.621559Z","shell.execute_reply.started":"2021-06-17T08:53:04.177814Z","shell.execute_reply":"2021-06-17T08:53:09.620432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.copytree(YOLOV5_DIR, YOLOV5_W_DIR) ","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:31:10.954262Z","iopub.execute_input":"2021-06-14T17:31:10.954563Z","iopub.status.idle":"2021-06-14T17:31:15.01288Z","shell.execute_reply.started":"2021-06-14T17:31:10.95453Z","shell.execute_reply":"2021-06-14T17:31:15.011528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **load csv file**","metadata":{}},{"cell_type":"code","source":"df_train_image_level = pd.read_csv(META_PATH)\ndf_train_image_level['path'] = df_train_image_level.apply(lambda row: INPUT_DIR+(row.path.split('/')[-1]), axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-17T08:57:45.401633Z","iopub.execute_input":"2021-06-17T08:57:45.402101Z","iopub.status.idle":"2021-06-17T08:57:45.521248Z","shell.execute_reply.started":"2021-06-17T08:57:45.402070Z","shell.execute_reply":"2021-06-17T08:57:45.520083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Object Detection with yolov5**","metadata":{}},{"cell_type":"markdown","source":"**train test split**","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(df_train_image_level, test_size=0.2, random_state=42)\n\ntrain_df = train_df.copy()\nvalid_df = valid_df.copy()\n\ntrain_df.loc[:, 'split'] = 'train'\nvalid_df.loc[:, 'split'] = 'valid'\n\ndf_train_image_level = pd.concat([train_df, valid_df]).reset_index(drop=True)\ndf_train_image_level.sample(4)","metadata":{"execution":{"iopub.status.busy":"2021-06-17T08:53:17.718491Z","iopub.execute_input":"2021-06-17T08:53:17.718889Z","iopub.status.idle":"2021-06-17T08:53:17.775204Z","shell.execute_reply.started":"2021-06-17T08:53:17.718848Z","shell.execute_reply":"2021-06-17T08:53:17.774088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**create dir split train valid and copy image for training**","metadata":{}},{"cell_type":"code","source":"[os.makedirs(dir, exist_ok=True) for dir in [TRAIN_IMAGES_DIR,\n                                             VAL_IMAGES_DIR,\n                                             TRAIN_LABELS_DIR,\n                                             VAL_LABELS_DIR]]\nfor i in tqdm(range(len(df_train_image_level))):\n    row = df_train_image_level.loc[i]\n    if os.path.exists(row.path):\n        if row.split == 'train':\n            shutil.copy(row.path, f'{TRAIN_IMAGES_DIR}{row.id}.jpg')\n        else:\n            shutil.copy(row.path, f'{VAL_IMAGES_DIR}{row.id}.jpg')","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:57:20.571521Z","iopub.execute_input":"2021-06-14T17:57:20.571802Z","iopub.status.idle":"2021-06-14T17:57:27.22015Z","shell.execute_reply.started":"2021-06-14T17:57:20.571774Z","shell.execute_reply":"2021-06-14T17:57:27.219194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**create data yaml yolov5**","metadata":{}},{"cell_type":"code","source":"data_yaml = dict(\n    train = f'../../../../../.{TRAIN_IMAGES_DIR}',\n    val = f'../../../../../.{VAL_IMAGES_DIR}',\n    nc = 2,\n    names = ['none','opacity']\n)\n\nwith open(YOLOV5_W_DIR+'data/data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:57:27.221531Z","iopub.execute_input":"2021-06-14T17:57:27.221845Z","iopub.status.idle":"2021-06-14T17:57:27.230513Z","shell.execute_reply.started":"2021-06-14T17:57:27.2218Z","shell.execute_reply":"2021-06-14T17:57:27.229541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**convert results .txt to df_train_image_level**","metadata":{}},{"cell_type":"code","source":"def get_bbox(row):\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row.label.split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l))\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []  \n            \n    return bboxes\n\ndef scale_bbox(row, bboxes):\n    # Get scaling factor\n    scale_x = IMG_SIZE/row.width\n    scale_y = IMG_SIZE/row.height\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        x = int(np.round(bbox[0]*scale_x, 4))\n        y = int(np.round(bbox[1]*scale_y, 4))\n        x1 = int(np.round(bbox[2]*(scale_x), 4))\n        y1= int(np.round(bbox[3]*scale_y, 4))\n\n        scaled_bboxes.append([x, y, x1, y1]) \n        \n    return scaled_bboxes\n\ndef get_yolo_format_bbox(img_w, img_h, bboxes):\n    yolo_boxes = []\n    for bbox in bboxes:\n        w = bbox[2] - bbox[0] \n        h = bbox[3] - bbox[1] \n        xc = bbox[0] + int(np.round(w/2)) \n        yc = bbox[1] + int(np.round(h/2)) \n        \n        yolo_boxes.append([f'{xc/img_w:.6f}',f'{yc/img_h:.6f}' ,f'{w/img_w:.6f}' ,f'{h/img_h:.6f}'])\n    \n    return yolo_boxes","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:57:27.232012Z","iopub.execute_input":"2021-06-14T17:57:27.232317Z","iopub.status.idle":"2021-06-14T17:57:27.261125Z","shell.execute_reply.started":"2021-06-14T17:57:27.232289Z","shell.execute_reply":"2021-06-14T17:57:27.260154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**create label txt files**","metadata":{}},{"cell_type":"code","source":"if not os.listdir(f'{TRAIN_LABELS_DIR}'):\n    for i in tqdm(range(len(df_train_image_level))):\n        row = df_train_image_level.loc[i]\n        img_id = row.id\n        split = row.split\n        label = row.image_level\n\n        if row.split=='train':\n            file_name = f'{TRAIN_LABELS_DIR}{row.id}.txt'\n        else:\n            file_name = f'{VAL_LABELS_DIR}{row.id}.txt'\n\n\n        if label=='opacity':\n            bboxes = get_bbox(row)\n            scale_bboxes = scale_bbox(row, bboxes)\n            yolo_bboxes = get_yolo_format_bbox(IMG_SIZE, IMG_SIZE, scale_bboxes)\n\n            with open(file_name, 'w') as f:\n                for bbox in yolo_bboxes:\n                    bbox = [1]+bbox\n                    bbox = [str(i) for i in bbox]\n                    bbox = ' '.join(bbox)\n                    f.write(bbox)\n                    f.write('\\n')","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:57:27.262535Z","iopub.execute_input":"2021-06-14T17:57:27.262877Z","iopub.status.idle":"2021-06-14T17:57:30.518748Z","shell.execute_reply.started":"2021-06-14T17:57:27.262823Z","shell.execute_reply":"2021-06-14T17:57:30.517721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**training**","metadata":{}},{"cell_type":"code","source":"%cd {YOLOV5_W_DIR}","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:57:30.51998Z","iopub.execute_input":"2021-06-14T17:57:30.520282Z","iopub.status.idle":"2021-06-14T17:57:30.532934Z","shell.execute_reply.started":"2021-06-14T17:57:30.520252Z","shell.execute_reply":"2021-06-14T17:57:30.530766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python train.py --img {IMG_SIZE} \\\n                 --batch-size {BATCH_SIZE} \\\n                 --epochs {EPOCHS} \\\n                 --data data.yaml \\\n                 --weights yolov5s.pt \\\n                 --save_period 1\\\n                 --project {WANDB_PROJECT_NAME}","metadata":{"execution":{"iopub.status.busy":"2021-06-14T17:57:30.534979Z","iopub.execute_input":"2021-06-14T17:57:30.535468Z","iopub.status.idle":"2021-06-14T18:04:43.397522Z","shell.execute_reply.started":"2021-06-14T17:57:30.535421Z","shell.execute_reply":"2021-06-14T18:04:43.39632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **ref**","metadata":{}},{"cell_type":"markdown","source":"\n* https://www.kaggle.com/xhlulu\n* https://www.kaggle.com/yujiariyasu\n* https://www.kaggle.com/ayuraj\n* https://www.kaggle.com/dschettler8845   \n....","metadata":{}}]}