{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-23T00:26:03.090442Z","iopub.execute_input":"2021-07-23T00:26:03.090907Z","iopub.status.idle":"2021-07-23T00:26:11.914061Z","shell.execute_reply.started":"2021-07-23T00:26:03.090847Z","shell.execute_reply":"2021-07-23T00:26:11.913012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pydicom\nimport glob\nfrom tqdm.notebook import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\nwarnings.filterwarnings('ignore')\ndataset_path = Path('../input/siim-covid19-detection')","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:26:34.561942Z","iopub.execute_input":"2021-07-23T00:26:34.562313Z","iopub.status.idle":"2021-07-23T00:26:35.876046Z","shell.execute_reply.started":"2021-07-23T00:26:34.56228Z","shell.execute_reply":"2021-07-23T00:26:35.875016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(dataset_path/'train_image_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:26:48.228834Z","iopub.execute_input":"2021-07-23T00:26:48.229243Z","iopub.status.idle":"2021-07-23T00:26:48.284423Z","shell.execute_reply.started":"2021-07-23T00:26:48.229203Z","shell.execute_reply":"2021-07-23T00:26:48.283345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd ../\n!mkdir tmp\n%cd tmp","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:26:51.161083Z","iopub.execute_input":"2021-07-23T00:26:51.161499Z","iopub.status.idle":"2021-07-23T00:26:51.941141Z","shell.execute_reply.started":"2021-07-23T00:26:51.161446Z","shell.execute_reply":"2021-07-23T00:26:51.940201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Download YOLOv5\n!git clone https://github.com/ultralytics/yolov5  # clone repo\n%cd yolov5\n# Install dependencies\n%pip install -qr requirements.txt  # install dependencies\n\n%cd ../\nimport torch\nprint(f\"Setup complete. Using torch {torch.__version__} ({torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else 'CPU'})\")","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:26:53.049282Z","iopub.execute_input":"2021-07-23T00:26:53.049814Z","iopub.status.idle":"2021-07-23T00:27:00.082042Z","shell.execute_reply.started":"2021-07-23T00:26:53.049771Z","shell.execute_reply":"2021-07-23T00:27:00.080799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install W&B \n!pip install -q --upgrade wandb\n# Login \nimport wandb\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:27:00.084224Z","iopub.execute_input":"2021-07-23T00:27:00.084733Z","iopub.status.idle":"2021-07-23T00:27:07.297936Z","shell.execute_reply.started":"2021-07-23T00:27:00.084682Z","shell.execute_reply":"2021-07-23T00:27:07.297112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Necessary/extra dependencies. \nimport os\nimport gc\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom shutil import copyfile\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\n#customize iPython writefile so we can write variables\nfrom IPython.core.magic import register_line_cell_magic\n\n@register_line_cell_magic\ndef writetemplate(line, cell):\n    with open(line, 'w') as f:\n        f.write(cell.format(**globals()))","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:27:10.701126Z","iopub.execute_input":"2021-07-23T00:27:10.701797Z","iopub.status.idle":"2021-07-23T00:27:10.707402Z","shell.execute_reply.started":"2021-07-23T00:27:10.701753Z","shell.execute_reply":"2021-07-23T00:27:10.706675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = 'input/siim-covid19-resized-to-256px-jpg/train/'\nIMG_SIZE = 256\nBATCH_SIZE = 16\nEPOCHS = 10","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:27:20.256841Z","iopub.execute_input":"2021-07-23T00:27:20.257498Z","iopub.status.idle":"2021-07-23T00:27:20.261175Z","shell.execute_reply.started":"2021-07-23T00:27:20.257442Z","shell.execute_reply":"2021-07-23T00:27:20.260484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Everything is done from /kaggle directory.\n%cd ../\n\n# Modify values in the id column\ndf['id'] = df.apply(lambda row: row.id.split('_')[0], axis=1)\n# Add absolute path\ndf['path'] = df.apply(lambda row: TRAIN_PATH+row.id+'.jpg', axis=1)\n# Get image level labels\ndf['image_level'] = df.apply(lambda row: row.label.split(' ')[0], axis=1)\n\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:27:33.045404Z","iopub.execute_input":"2021-07-23T00:27:33.046014Z","iopub.status.idle":"2021-07-23T00:27:33.309588Z","shell.execute_reply.started":"2021-07-23T00:27:33.045978Z","shell.execute_reply":"2021-07-23T00:27:33.308535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load meta.csv file\n# Original dimensions are required to scale the bounding box coordinates appropriately.\nmeta_df = pd.read_csv('input/siim-covid19-resized-to-256px-jpg/meta.csv')\ntrain_meta_df = meta_df.loc[meta_df.split == 'train']\ntrain_meta_df = train_meta_df.drop('split', axis=1)\ntrain_meta_df.columns = ['id', 'dim0', 'dim1']\n\ntrain_meta_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:29:27.816763Z","iopub.execute_input":"2021-07-23T00:29:27.81771Z","iopub.status.idle":"2021-07-23T00:29:27.866544Z","shell.execute_reply.started":"2021-07-23T00:29:27.817653Z","shell.execute_reply":"2021-07-23T00:29:27.865521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge both the dataframes\ndf = df.merge(train_meta_df, on='id',how=\"left\")\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:29:53.401261Z","iopub.execute_input":"2021-07-23T00:29:53.401703Z","iopub.status.idle":"2021-07-23T00:29:53.438327Z","shell.execute_reply.started":"2021-07-23T00:29:53.401664Z","shell.execute_reply":"2021-07-23T00:29:53.437243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create train and validation split.\ntrain_df, valid_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df.image_level.values)\n\ntrain_df.loc[:, 'split'] = 'train'\nvalid_df.loc[:, 'split'] = 'valid'\n\ndf = pd.concat([train_df, valid_df]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:30:00.960827Z","iopub.execute_input":"2021-07-23T00:30:00.961202Z","iopub.status.idle":"2021-07-23T00:30:00.999124Z","shell.execute_reply.started":"2021-07-23T00:30:00.961171Z","shell.execute_reply":"2021-07-23T00:30:00.997751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('tmp/covid/images/train', exist_ok=True)\nos.makedirs('tmp/covid/images/valid', exist_ok=True)\n\nos.makedirs('tmp/covid/labels/train', exist_ok=True)\nos.makedirs('tmp/covid/labels/valid', exist_ok=True)\n\n! ls tmp/covid/images","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:30:06.965596Z","iopub.execute_input":"2021-07-23T00:30:06.965956Z","iopub.status.idle":"2021-07-23T00:30:07.761761Z","shell.execute_reply.started":"2021-07-23T00:30:06.965927Z","shell.execute_reply":"2021-07-23T00:30:07.760491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Move the images to relevant split folder.\nfor i in tqdm(range(len(df))):\n    row = df.loc[i]\n    if row.split == 'train':\n        copyfile(row.path, f'tmp/covid/images/train/{row.id}.jpg')\n    else:\n        copyfile(row.path, f'tmp/covid/images/valid/{row.id}.jpg')","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:30:12.305254Z","iopub.execute_input":"2021-07-23T00:30:12.305702Z","iopub.status.idle":"2021-07-23T00:30:57.406633Z","shell.execute_reply.started":"2021-07-23T00:30:12.305665Z","shell.execute_reply":"2021-07-23T00:30:57.405324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create .yaml file \nimport yaml\n\ndata_yaml = dict(\n    train = '../covid/images/train',\n    val = '../covid/images/valid',\n    nc = 2,\n    names = ['none', 'opacity']\n)\n\n# Note that I am creating the file in the yolov5/data/ directory.\nwith open('tmp/yolov5/data/data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=True)\n    \n%cat tmp/yolov5/data/data.yaml","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:30:57.40825Z","iopub.execute_input":"2021-07-23T00:30:57.408571Z","iopub.status.idle":"2021-07-23T00:30:58.192577Z","shell.execute_reply.started":"2021-07-23T00:30:57.408538Z","shell.execute_reply":"2021-07-23T00:30:58.191454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the raw bounding box by parsing the row value of the label column.\n# Ref: https://www.kaggle.com/yujiariyasu/plot-3positive-classes\ndef get_bbox(row):\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row.label.split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l))\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []  \n            \n    return bboxes\n\n# Scale the bounding boxes according to the size of the resized image. \ndef scale_bbox(row, bboxes):\n    # Get scaling factor\n    scale_x = IMG_SIZE/row.dim1\n    scale_y = IMG_SIZE/row.dim0\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        x = int(np.round(bbox[0]*scale_x, 4))\n        y = int(np.round(bbox[1]*scale_y, 4))\n        x1 = int(np.round(bbox[2]*(scale_x), 4))\n        y1= int(np.round(bbox[3]*scale_y, 4))\n\n        scaled_bboxes.append([x, y, x1, y1]) # xmin, ymin, xmax, ymax\n        \n    return scaled_bboxes\n\n# Convert the bounding boxes in YOLO format.\ndef get_yolo_format_bbox(img_w, img_h, bboxes):\n    yolo_boxes = []\n    for bbox in bboxes:\n        w = bbox[2] - bbox[0] # xmax - xmin\n        h = bbox[3] - bbox[1] # ymax - ymin\n        xc = bbox[0] + int(np.round(w/2)) # xmin + width/2\n        yc = bbox[1] + int(np.round(h/2)) # ymin + height/2\n        \n        yolo_boxes.append([xc/img_w, yc/img_h, w/img_w, h/img_h]) # x_center y_center width height\n    \n    return yolo_boxes","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:30:58.196304Z","iopub.execute_input":"2021-07-23T00:30:58.196662Z","iopub.status.idle":"2021-07-23T00:30:58.207786Z","shell.execute_reply.started":"2021-07-23T00:30:58.196629Z","shell.execute_reply":"2021-07-23T00:30:58.206578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare the txt files for bounding box\nfor i in tqdm(range(len(df))):\n    row = df.loc[i]\n    # Get image id\n    img_id = row.id\n    # Get split\n    split = row.split\n    # Get image-level label\n    label = row.image_level\n    \n    if row.split=='train':\n        file_name = f'tmp/covid/labels/train/{row.id}.txt'\n    else:\n        file_name = f'tmp/covid/labels/valid/{row.id}.txt'\n        \n    \n    if label=='opacity':\n        # Get bboxes\n        bboxes = get_bbox(row)\n        # Scale bounding boxes\n        scale_bboxes = scale_bbox(row, bboxes)\n        # Format for YOLOv5\n        yolo_bboxes = get_yolo_format_bbox(IMG_SIZE, IMG_SIZE, scale_bboxes)\n        \n        with open(file_name, 'w') as f:\n            for bbox in yolo_bboxes:\n                bbox = [1]+bbox\n                bbox = [str(i) for i in bbox]\n                bbox = ' '.join(bbox)\n                f.write(bbox)\n                f.write('\\n')","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:30:58.209994Z","iopub.execute_input":"2021-07-23T00:30:58.210504Z","iopub.status.idle":"2021-07-23T00:31:01.037586Z","shell.execute_reply.started":"2021-07-23T00:30:58.21043Z","shell.execute_reply":"2021-07-23T00:31:01.036598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd tmp/yolov5/","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:31:01.038822Z","iopub.execute_input":"2021-07-23T00:31:01.039119Z","iopub.status.idle":"2021-07-23T00:31:01.047927Z","shell.execute_reply.started":"2021-07-23T00:31:01.039089Z","shell.execute_reply":"2021-07-23T00:31:01.046625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python train.py --img {IMG_SIZE} \\\n                 --batch {BATCH_SIZE} \\\n                 --epochs {EPOCHS} \\\n                 --data data.yaml \\\n                 --weights yolov5s.pt \\\n                 --save_period 1\\\n                 --project kaggle-siim-covid","metadata":{"execution":{"iopub.status.busy":"2021-07-23T00:49:15.238658Z","iopub.execute_input":"2021-07-23T00:49:15.239195Z","iopub.status.idle":"2021-07-23T00:49:16.127549Z","shell.execute_reply.started":"2021-07-23T00:49:15.239074Z","shell.execute_reply":"2021-07-23T00:49:16.126558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = '/kaggle/input/siim-covid19-resized-to-256px-jpg/test/' # absolute path","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PATH = 'kaggle-siim-covid/exp/weights/best.pt'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights {MODEL_PATH} \\\n                  --source {TEST_PATH} \\\n                  --img {IMG_SIZE} \\\n                  --conf 0.281 \\\n                  --iou-thres 0.5 \\\n                  --max-det 3 \\\n                  --save-txt \\\n                  --save-conf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PRED_PATH = 'runs/detect/exp3/labels'\n!ls {PRED_PATH}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize predicted coordinates.\n%cat runs/detect/exp3/labels/ba91d37ee459.txt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_files = os.listdir(PRED_PATH)\nprint('Number of test images predicted as opaque: ', len(prediction_files))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The submisison requires xmin, ymin, xmax, ymax format. \n# YOLOv5 returns x_center, y_center, width, height\ndef correct_bbox_format(bboxes):\n    correct_bboxes = []\n    for b in bboxes:\n        xc, yc = int(np.round(b[0]*IMG_SIZE)), int(np.round(b[1]*IMG_SIZE))\n        w, h = int(np.round(b[2]*IMG_SIZE)), int(np.round(b[3]*IMG_SIZE))\n\n        xmin = xc - int(np.round(w/2))\n        xmax = xc + int(np.round(w/2))\n        ymin = yc - int(np.round(h/2))\n        ymax = yc + int(np.round(h/2))\n        \n        correct_bboxes.append([xmin, xmax, ymin, ymax])\n        \n    return correct_bboxes\n\n# Read the txt file generated by YOLOv5 during inference and extract \n# confidence and bounding box coordinates.\ndef get_conf_bboxes(file_path):\n    confidence = []\n    bboxes = []\n    with open(file_path, 'r') as file:\n        for line in file:\n            preds = line.strip('\\n').split(' ')\n            preds = list(map(float, preds))\n            confidence.append(preds[-1])\n            bboxes.append(preds[1:-1])\n    return confidence, bboxes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the submisison file\nsub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nsub_df.tail()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prediction loop for submission\npredictions = []\n\nfor i in tqdm(range(len(sub_df))):\n    row = sub_df.loc[i]\n    id_name = row.id.split('_')[0]\n    id_level = row.id.split('_')[-1]\n    \n    if id_level == 'study':\n        # do study-level classification\n        predictions.append(\"Negative 1 0 0 1 1\") # dummy prediction\n        \n    elif id_level == 'image':\n        # we can do image-level classification here.\n        # also we can rely on the object detector's classification head.\n        # for this example submisison we will use YOLO's classification head. \n        # since we already ran the inference we know which test images belong to opacity.\n        if f'{id_name}.txt' in prediction_files:\n            # opacity label\n            confidence, bboxes = get_conf_bboxes(f'{PRED_PATH}/{id_name}.txt')\n            bboxes = correct_bbox_format(bboxes)\n            pred_string = ''\n            for j, conf in enumerate(confidence):\n                pred_string += f'opacity {conf} ' + ' '.join(map(str, bboxes[j])) + ' '\n            predictions.append(pred_string[:-1]) \n        else:\n            predictions.append(\"None 1 0 0 1 1\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['PredictionString'] = predictions\nsub_df.to_csv('submission.csv', index=False)\nsub_df.tail()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    ","metadata":{},"execution_count":null,"outputs":[]}]}