{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%cd ../\n!mkdir tmp\n%cd tmp","metadata":{"execution":{"iopub.status.busy":"2021-07-16T23:59:57.447762Z","iopub.execute_input":"2021-07-16T23:59:57.44844Z","iopub.status.idle":"2021-07-16T23:59:58.14454Z","shell.execute_reply.started":"2021-07-16T23:59:57.448311Z","shell.execute_reply":"2021-07-16T23:59:58.143639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Download YOLOv5\n!git clone https://github.com/ultralytics/yolov5  # clone repo\n%cd yolov5\n# Install dependencies\n%pip install -qr requirements.txt  # install dependencies\n\n%cd ../\nimport torch\nprint(f\"Setup complete. Using torch {torch.__version__} ({torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else 'CPU'})\")","metadata":{"execution":{"iopub.status.busy":"2021-07-16T23:59:58.146233Z","iopub.execute_input":"2021-07-16T23:59:58.146586Z","iopub.status.idle":"2021-07-17T00:00:10.727754Z","shell.execute_reply.started":"2021-07-16T23:59:58.146548Z","shell.execute_reply":"2021-07-17T00:00:10.7269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install W&B \n!pip install -q --upgrade wandb\n# Login \nimport wandb\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:10.729439Z","iopub.execute_input":"2021-07-17T00:00:10.729696Z","iopub.status.idle":"2021-07-17T00:00:33.893664Z","shell.execute_reply.started":"2021-07-17T00:00:10.72967Z","shell.execute_reply":"2021-07-17T00:00:33.892841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Necessary/extra dependencies. \nimport os\nimport gc\nimport cv2 \nimport wandb\nimport shutil\nimport numpy as np\nimport pandas as pd\npd.set_option('mode.chained_assignment', None)\nfrom tqdm import tqdm\nfrom shutil import copyfile\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold\n\n#customize iPython writefile so we can write variables\nfrom IPython.core.magic import register_line_cell_magic\n\n@register_line_cell_magic\ndef writetemplate(line, cell):\n    with open(line, 'w') as f:\n        f.write(cell.format(**globals()))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:33.895416Z","iopub.execute_input":"2021-07-17T00:00:33.895807Z","iopub.status.idle":"2021-07-17T00:00:34.913606Z","shell.execute_reply.started":"2021-07-17T00:00:33.895768Z","shell.execute_reply":"2021-07-17T00:00:34.912498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = '../input/siim-covid19-resized-to-256px-png/train'\nIMG_SIZE = 256\nBATCH_SIZE = 32\nEPOCHS = 100\nUSE_FOLD = False\nSEED = 42\nNUM_FOLD = 5","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:34.91497Z","iopub.execute_input":"2021-07-17T00:00:34.915339Z","iopub.status.idle":"2021-07-17T00:00:34.923447Z","shell.execute_reply.started":"2021-07-17T00:00:34.915303Z","shell.execute_reply":"2021-07-17T00:00:34.922555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# IMAGE LEVEL\n\n# Load image level csv file\ndf = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\n# Load study level csv file\nlabel_df = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\n\n# Modify values in the id column\ndf['id'] = df.apply(lambda row: row.id.split('_')[0], axis=1)\n# Add absolute path\n# df['path'] = df.apply(lambda row: TRAIN_PATH+row.id+'.jpg', axis=1)\n# Get image level labels\ndef image_level(row):\n    label = row.label.split(' ')[0]\n    if label == 'opacity': return 1\n    else: return 0\n\ndf['image_level'] = df.apply(lambda row: image_level(row), axis=1)\n\n# STUDY LEVEL\n\n# Modify values in the id column\nlabel_df['id'] = label_df.apply(lambda row: row.id.split('_')[0], axis=1)\n# Rename the column id with StudyInstanceUID\nlabel_df.columns = ['StudyInstanceUID', 'Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']\n\n# Label encode study-level labels\nlabels = label_df[['Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']].values\nlabels = np.argmax(labels, axis=1)\nlabel_df['study_level'] = labels\n\n# ORIGINAL DIMENSION\n\n# Load meta.csv file\nmeta_df = pd.read_csv('../input/siim-covid19-resized-to-256px-png/meta.csv')\ntrain_meta_df = meta_df.loc[meta_df.split == 'train']\ntrain_meta_df = train_meta_df.drop('split', axis=1)\ntrain_meta_df.columns = ['id', 'dim0', 'dim1']\n\n# Merge image-level and study-level\ndf = df.merge(label_df, on='StudyInstanceUID',how=\"left\")\n# Merge with meta_df\ndf = df.merge(train_meta_df, on='id',how=\"left\")\n\n# Write as csv file\ndf.to_csv('_image_study_total.csv', index=False)\n\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:34.925203Z","iopub.execute_input":"2021-07-17T00:00:34.925738Z","iopub.status.idle":"2021-07-17T00:00:35.434575Z","shell.execute_reply.started":"2021-07-17T00:00:34.925696Z","shell.execute_reply":"2021-07-17T00:00:35.433796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'train_fold.csv' in os.listdir(os.getcwd()) and USE_FOLD:\n    df = pd.read_csv('train_fold.csv')\nelse:\n    df = pd.read_csv('_image_study_total.csv')\n    df['path'] = df.apply(lambda row: f'../input/siim-covid19-resized-to-256px-png/train/{row.id}.png', axis=1)\n    \n    # Group by Study Ids and remove images that are \"assumed\" to be mislabeled\n    for grp_df in df.groupby('StudyInstanceUID'):\n        grp_id, grp_df = grp_df[0], grp_df[1]\n        if len(grp_df) == 1:\n            pass\n        else:\n            for i in range(len(grp_df)):\n                row = grp_df.loc[grp_df.index.values[i]]\n                if row.study_level > 0 and row.boxes is np.nan:\n                    df = df.drop(grp_df.index.values[i])\n                    \n    print('total number of images: ', len(df))\n    \n    # Create train and validation split.\n    df = df.drop('boxes', axis=1).reset_index()\n    Fold = StratifiedKFold(n_splits=NUM_FOLD, shuffle=True, random_state=SEED)\n    for n, (train_index, val_index) in enumerate(Fold.split(df, df['image_level'])):\n        df.loc[val_index, 'fold'] = int(n)\n    df['fold'] = df['fold'].astype(int)\n\n    df.to_csv('train_fold.csv', index=False)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:35.435864Z","iopub.execute_input":"2021-07-17T00:00:35.436204Z","iopub.status.idle":"2021-07-17T00:00:36.169752Z","shell.execute_reply.started":"2021-07-17T00:00:35.436169Z","shell.execute_reply":"2021-07-17T00:00:36.168797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pwd","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:36.172483Z","iopub.execute_input":"2021-07-17T00:00:36.172891Z","iopub.status.idle":"2021-07-17T00:00:36.850679Z","shell.execute_reply.started":"2021-07-17T00:00:36.172852Z","shell.execute_reply":"2021-07-17T00:00:36.849728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -la","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:36.85277Z","iopub.execute_input":"2021-07-17T00:00:36.853146Z","iopub.status.idle":"2021-07-17T00:00:37.541998Z","shell.execute_reply.started":"2021-07-17T00:00:36.853093Z","shell.execute_reply":"2021-07-17T00:00:37.540914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_FOLD:\n    pass\nelse:\n    # Remove existing dirs\n    for fold in range(NUM_FOLD):\n        # Prepare train and valid df\n        train_df = df.loc[df.fold != fold].reset_index(drop=True)\n        valid_df = df.loc[df.fold == fold].reset_index(drop=True)\n        \n        try:\n            shutil.rmtree(f'dataset_folds_{fold}/images')\n            shutil.rmtree(f'dataset_folds_{fold}/labels')\n        except:\n            print('No dirs')\n\n        # Make new dirs\n        os.makedirs(f'dataset_folds_{fold}/images/train', exist_ok=True)\n        os.makedirs(f'dataset_folds_{fold}/images/valid', exist_ok=True)\n        os.makedirs(f'dataset_folds_{fold}/labels/train', exist_ok=True)\n        os.makedirs(f'dataset_folds_{fold}/labels/valid', exist_ok=True)\n\n        # Move the images to relevant split folder.\n        for i in tqdm(range(len(train_df))):\n            row = train_df.loc[i]\n            copyfile(row.path, f'dataset_folds_{fold}/images/train/{row.id}.png')\n            \n        for i in tqdm(range(len(valid_df))):\n            row = valid_df.loc[i]\n            copyfile(row.path, f'dataset_folds_{fold}/images/valid/{row.id}.png')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:00:37.545809Z","iopub.execute_input":"2021-07-17T00:00:37.546141Z","iopub.status.idle":"2021-07-17T00:01:32.419412Z","shell.execute_reply.started":"2021-07-17T00:00:37.546096Z","shell.execute_reply":"2021-07-17T00:01:32.418524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:01:32.420673Z","iopub.execute_input":"2021-07-17T00:01:32.421023Z","iopub.status.idle":"2021-07-17T00:01:33.072197Z","shell.execute_reply.started":"2021-07-17T00:01:32.420987Z","shell.execute_reply":"2021-07-17T00:01:33.071292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create .yaml file \nimport yaml\n\nfor fold in range(NUM_FOLD):\n    data_yaml = dict(\n        train = f'../dataset_folds_{fold}/images/train',\n        val = f'../dataset_folds_{fold}/images/valid',\n        nc = 2,\n        names = ['none', 'opacity']\n    )\n\n    # Note that I am creating the file in the yolov5/data/ directory.\n    with open(f'yolov5/data/data_fold_{fold}.yaml', 'w') as outfile:\n        yaml.dump(data_yaml, outfile, default_flow_style=True)\n    \n%cat yolov5/data/data_fold_0.yaml","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:01:33.075363Z","iopub.execute_input":"2021-07-17T00:01:33.075636Z","iopub.status.idle":"2021-07-17T00:01:33.7472Z","shell.execute_reply.started":"2021-07-17T00:01:33.075609Z","shell.execute_reply":"2021-07-17T00:01:33.746276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the raw bounding box by parsing the row value of the label column.\n# Ref: https://www.kaggle.com/yujiariyasu/plot-3positive-classes\ndef get_bbox(row):\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row.label.split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l))\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []  \n            \n    return bboxes\n\n# Scale the bounding boxes according to the size of the resized image. \ndef scale_bbox(row, bboxes):\n    # Get scaling factor\n    scale_x = IMG_SIZE/row.dim1\n    scale_y = IMG_SIZE/row.dim0\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        x = int(np.round(bbox[0]*scale_x, 4))\n        y = int(np.round(bbox[1]*scale_y, 4))\n        x1 = int(np.round(bbox[2]*(scale_x), 4))\n        y1= int(np.round(bbox[3]*scale_y, 4))\n\n        scaled_bboxes.append([x, y, x1, y1]) # xmin, ymin, xmax, ymax\n        \n    return scaled_bboxes\n\n# Convert the bounding boxes in YOLO format.\ndef get_yolo_format_bbox(img_w, img_h, bboxes):\n    yolo_boxes = []\n    for bbox in bboxes:\n        w = bbox[2] - bbox[0] # xmax - xmin\n        h = bbox[3] - bbox[1] # ymax - ymin\n        xc = bbox[0] + int(np.round(w/2)) # xmin + width/2\n        yc = bbox[1] + int(np.round(h/2)) # ymin + height/2\n        \n        yolo_boxes.append([xc/img_w, yc/img_h, w/img_w, h/img_h]) # x_center y_center width height\n    \n    return yolo_boxes","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:01:33.748853Z","iopub.execute_input":"2021-07-17T00:01:33.749218Z","iopub.status.idle":"2021-07-17T00:01:33.762258Z","shell.execute_reply.started":"2021-07-17T00:01:33.749177Z","shell.execute_reply":"2021-07-17T00:01:33.760018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_bbox_files(tmp_df, fold_num, split):\n    path = f'dataset_folds_{fold}/labels/{split}'\n    for i in tqdm(range(len(tmp_df))):\n        row = tmp_df.loc[i]\n        # Get image id\n        img_id = row.id\n        # Get image-level label\n        label = row.image_level\n\n        file_name = f'{path}/{img_id}.txt'\n\n        if label==1:\n            # Get bboxes\n            bboxes = get_bbox(row)\n            # Scale bounding boxes\n            scale_bboxes = scale_bbox(row, bboxes)\n            # Format for YOLOv5\n            yolo_bboxes = get_yolo_format_bbox(IMG_SIZE, IMG_SIZE, scale_bboxes)\n\n            with open(file_name, 'w') as f:\n                for bbox in yolo_bboxes:\n                    bbox = [1]+bbox\n                    bbox = [str(i) for i in bbox]\n                    bbox = ' '.join(bbox)\n                    f.write(bbox)\n                    f.write('\\n')\n\nif USE_FOLD:\n    pass\nelse:\n    # Prepare the txt files for bounding box\n    for fold in range(NUM_FOLD):\n        # Prepare train and valid df\n        train_df = df.loc[df.fold != fold].reset_index(drop=True)\n        valid_df = df.loc[df.fold == fold].reset_index(drop=True)\n        \n        # prepare label for train\n        write_bbox_files(train_df, fold, 'train')\n        # prepare label for valid\n        write_bbox_files(valid_df, fold, 'valid')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:01:33.763618Z","iopub.execute_input":"2021-07-17T00:01:33.76416Z","iopub.status.idle":"2021-07-17T00:01:47.584569Z","shell.execute_reply.started":"2021-07-17T00:01:33.764116Z","shell.execute_reply":"2021-07-17T00:01:47.583628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd yolov5","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:01:47.586198Z","iopub.execute_input":"2021-07-17T00:01:47.586619Z","iopub.status.idle":"2021-07-17T00:01:47.592673Z","shell.execute_reply.started":"2021-07-17T00:01:47.58658Z","shell.execute_reply":"2021-07-17T00:01:47.591908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold in range(NUM_FOLD):    \n    print('FOLD NUMBER: ', fold)\n    !python train.py --img {IMG_SIZE} \\\n                     --batch {BATCH_SIZE} \\\n                     --epochs {1} \\\n                     --data data_fold_{fold}.yaml \\\n                     --weights yolov5s.pt \\\n                     --save_period 10\\\n                     --project yolov5-covid19-folds\\\n                     --name yolov5s-e-100-img-256-fold-{fold}\n    print('###########################################################################################\\n')","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:01:47.594117Z","iopub.execute_input":"2021-07-17T00:01:47.594747Z","iopub.status.idle":"2021-07-17T00:12:11.98478Z","shell.execute_reply.started":"2021-07-17T00:01:47.59471Z","shell.execute_reply":"2021-07-17T00:12:11.983823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls yolov5-covid19-folds","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:12:11.986455Z","iopub.execute_input":"2021-07-17T00:12:11.986816Z","iopub.status.idle":"2021-07-17T00:12:12.637604Z","shell.execute_reply.started":"2021-07-17T00:12:11.986777Z","shell.execute_reply":"2021-07-17T00:12:12.636695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_WEIGHTS = [\n    'yolov5-covid19-folds/yolov5s-e-100-img-256-fold-0/weights/best.pt',\n    'yolov5-covid19-folds/yolov5s-e-100-img-256-fold-1/weights/best.pt',\n    'yolov5-covid19-folds/yolov5s-e-100-img-256-fold-2/weights/best.pt',\n    'yolov5-covid19-folds/yolov5s-e-100-img-256-fold-3/weights/best.pt',\n    'yolov5-covid19-folds/yolov5s-e-100-img-256-fold-4/weights/best.pt',\n]\n\nSOURCES = [\n    '../dataset_folds_0/images/valid',\n    '../dataset_folds_1/images/valid',\n    '../dataset_folds_2/images/valid',\n    '../dataset_folds_3/images/valid',\n    '../dataset_folds_4/images/valid',\n]\n\nCONFIDENCE = [\n    0.269, 0.268, 0.209, 0.179, 0.308\n]","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:14:13.323665Z","iopub.execute_input":"2021-07-17T00:14:13.324034Z","iopub.status.idle":"2021-07-17T00:14:13.330691Z","shell.execute_reply.started":"2021-07-17T00:14:13.323999Z","shell.execute_reply":"2021-07-17T00:14:13.32974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold in range(NUM_FOLD):\n    print('FOLD NUMBER: ', fold)\n    \n    !python detect.py --weights {MODEL_WEIGHTS[fold]} \\\n                      --source {SOURCES[fold]} \\\n                      --img {IMG_SIZE} \\\n                      --conf {CONFIDENCE[fold]} \\\n                      --iou-thres 0.5 \\\n                      --max-det 3 \\\n                      --name infer_fold_{fold}\\\n                      --save-txt \\\n                      --save-conf\\\n                      --nosave\n    \n    print('###########################################################################################\\n')    ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T00:14:33.49561Z","iopub.execute_input":"2021-07-17T00:14:33.495932Z","iopub.status.idle":"2021-07-17T00:17:25.345987Z","shell.execute_reply.started":"2021-07-17T00:14:33.495902Z","shell.execute_reply":"2021-07-17T00:17:25.34441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}