{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-09T14:04:52.689290Z","iopub.execute_input":"2021-08-09T14:04:52.689727Z","iopub.status.idle":"2021-08-09T14:04:52.700809Z","shell.execute_reply.started":"2021-08-09T14:04:52.689634Z","shell.execute_reply":"2021-08-09T14:04:52.699382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir tmp\n%cd tmp","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:52.908110Z","iopub.execute_input":"2021-08-09T14:04:52.908515Z","iopub.status.idle":"2021-08-09T14:04:53.593765Z","shell.execute_reply.started":"2021-08-09T14:04:52.908456Z","shell.execute_reply":"2021-08-09T14:04:53.592887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install torch==1.5.1+cu101 torchvision==0.6.1+cu101 -f https://download.pytorch.org/whl/torch_stable.html\n!pip install numpy==1.17\n!pip install PyYAML==5.3.1\n!pip install git+https://github.com/cocodataset/cocoapi.git#subdirectory=PythonAPI","metadata":{"execution":{"iopub.status.busy":"2021-08-09T12:00:21.182925Z","iopub.execute_input":"2021-08-09T12:00:21.183300Z","iopub.status.idle":"2021-08-09T12:00:34.360781Z","shell.execute_reply.started":"2021-08-09T12:00:21.183259Z","shell.execute_reply":"2021-08-09T12:00:34.359887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/NVIDIA/apex && cd apex && pip install -v --no-cache-dir --global-option=\"--cpp_ext\" --global-option=\"--cuda_ext\" . --user && cd .. && rm -rf apex","metadata":{"execution":{"iopub.status.busy":"2021-08-09T08:06:48.524042Z","iopub.execute_input":"2021-08-09T08:06:48.524403Z","iopub.status.idle":"2021-08-09T08:06:51.918422Z","shell.execute_reply.started":"2021-08-09T08:06:48.524364Z","shell.execute_reply":"2021-08-09T08:06:51.917461Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cloning the github repository.","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nfrom tqdm import tqdm\nimport numpy as np\nimport json\nimport urllib\nimport PIL.Image as Image\nimport cv2\nimport torch\nimport torchvision\nfrom IPython.display import display\nfrom sklearn.model_selection import train_test_split\nimport seaborn as sns\nfrom pylab import rcParams\nimport matplotlib.pyplot as plt\nfrom matplotlib import rc\n%matplotlib inline\n%config InlineBackend.figure_format='retina'\nsns.set(style='whitegrid', palette='muted', font_scale=1.2)\nrcParams['figure.figsize'] = 16, 10\nnp.random.seed(42)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:56.665298Z","iopub.execute_input":"2021-08-09T14:04:56.665637Z","iopub.status.idle":"2021-08-09T14:04:59.147989Z","shell.execute_reply.started":"2021-08-09T14:04:56.665604Z","shell.execute_reply":"2021-08-09T14:04:59.147166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/ultralytics/yolov5  # clone repo\n%cd yolov5\n# Install dependencies\n%pip install -qr requirements.txt  # install dependencies\n\n%cd ../\nimport torch\nfrom IPython.display import Image, clear_output  # to display images\n\nclear_output()\nprint(f\"Setup complete. Using torch {torch.__version__} ({torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else 'CPU'})\")","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:59.149309Z","iopub.execute_input":"2021-08-09T14:04:59.149639Z","iopub.status.idle":"2021-08-09T14:05:09.919562Z","shell.execute_reply.started":"2021-08-09T14:04:59.149606Z","shell.execute_reply":"2021-08-09T14:05:09.918401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"WANDB login (model artifacts will be stored on wandb account)","metadata":{}},{"cell_type":"code","source":"!pip install -q --upgrade wandb\n# Login \nimport wandb\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:09.921759Z","iopub.execute_input":"2021-08-09T14:05:09.922381Z","iopub.status.idle":"2021-08-09T14:05:25.490290Z","shell.execute_reply.started":"2021-08-09T14:05:09.922338Z","shell.execute_reply":"2021-08-09T14:05:25.489253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom shutil import copyfile\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\n#customize iPython writefile so we can write variables\nfrom IPython.core.magic import register_line_cell_magic\n\n@register_line_cell_magic\ndef writetemplate(line, cell):\n    with open(line, 'w') as f:\n        f.write(cell.format(**globals()))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:25.492045Z","iopub.execute_input":"2021-08-09T14:05:25.492415Z","iopub.status.idle":"2021-08-09T14:05:25.498373Z","shell.execute_reply.started":"2021-08-09T14:05:25.492364Z","shell.execute_reply":"2021-08-09T14:05:25.497334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Change this TRAIN_PATH as per convinience. It is path to modified training dataset which ","metadata":{}},{"cell_type":"code","source":"TRAIN_PATH = '/kaggle/input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/train/'\nIMG_SIZE = 640\nBATCH_SIZE = 16   # 16 if yolov5x\nEPOCHS = 30","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:38.620719Z","iopub.execute_input":"2021-08-08T07:49:38.621066Z","iopub.status.idle":"2021-08-08T07:49:38.627866Z","shell.execute_reply.started":"2021-08-08T07:49:38.621039Z","shell.execute_reply":"2021-08-08T07:49:38.627002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Modifying train_csv(because images are resized), adding absolute path, splitting data ","metadata":{}},{"cell_type":"markdown","source":"loading train_image_level.csv and making some modifications.\n\nAdding absolute path of images, adding image level labels","metadata":{}},{"cell_type":"code","source":"%cd ../\n%cd ../\n# Load image level csv file\ndf = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv')\n\n# Modify values in the id column\ndf['id'] = df.apply(lambda row: row.id.split('_')[0], axis=1)\n# Add absolute path\ndf['path'] = df.apply(lambda row: TRAIN_PATH+row.id+'.jpg', axis=1)\n# Get image level labels\ndf['image_level'] = df.apply(lambda row: row.label.split(' ')[0], axis=1)\n\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:38.629402Z","iopub.execute_input":"2021-08-08T07:49:38.631308Z","iopub.status.idle":"2021-08-08T07:49:39.013573Z","shell.execute_reply.started":"2021-08-08T07:49:38.631273Z","shell.execute_reply":"2021-08-08T07:49:39.012397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"meta_df is stored in modified dataset folder. It specified dimension ratio by which images are shrunk. ","metadata":{}},{"cell_type":"code","source":"meta_df = pd.read_csv('/kaggle/input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/meta_sz_640.csv')\ntrain_meta_df = meta_df.loc[meta_df.split == 'train']\ntrain_meta_df = train_meta_df.drop('split', axis=1)\ntrain_meta_df.columns = ['id', 'dim0', 'dim1']\n\ntrain_meta_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.015245Z","iopub.execute_input":"2021-08-08T07:49:39.015665Z","iopub.status.idle":"2021-08-08T07:49:39.060702Z","shell.execute_reply.started":"2021-08-08T07:49:39.015622Z","shell.execute_reply":"2021-08-08T07:49:39.059745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge both the dataframes\ndf = df.merge(train_meta_df, on='id',how=\"left\")\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.062106Z","iopub.execute_input":"2021-08-08T07:49:39.062656Z","iopub.status.idle":"2021-08-08T07:49:39.089654Z","shell.execute_reply.started":"2021-08-08T07:49:39.062613Z","shell.execute_reply":"2021-08-08T07:49:39.088927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create train and validation split.\ntrain_df, valid_df = train_test_split(df, test_size=0.15, random_state=42, stratify=df.image_level.values)\n\ntrain_df.loc[:, 'split'] = 'train'\nvalid_df.loc[:, 'split'] = 'valid'\n\ndf = pd.concat([train_df, valid_df]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.092376Z","iopub.execute_input":"2021-08-08T07:49:39.092639Z","iopub.status.idle":"2021-08-08T07:49:39.127537Z","shell.execute_reply.started":"2021-08-08T07:49:39.092614Z","shell.execute_reply":"2021-08-08T07:49:39.125679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Size of dataset: {len(df)}, training images: {len(train_df)}. validation images: {len(valid_df)}')","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.12912Z","iopub.execute_input":"2021-08-08T07:49:39.129592Z","iopub.status.idle":"2021-08-08T07:49:39.13578Z","shell.execute_reply.started":"2021-08-08T07:49:39.129547Z","shell.execute_reply":"2021-08-08T07:49:39.134376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nos.makedirs('working/tmp/covid/images/train', exist_ok=True)\nos.makedirs('working/tmp/covid/images/valid', exist_ok=True)\n\nos.makedirs('working/tmp/covid/labels/train', exist_ok=True)\nos.makedirs('working/tmp/covid/labels/valid', exist_ok=True)\n\n! ls working/tmp/covid/images","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.138302Z","iopub.execute_input":"2021-08-08T07:49:39.138725Z","iopub.status.idle":"2021-08-08T07:49:39.871775Z","shell.execute_reply.started":"2021-08-08T07:49:39.138692Z","shell.execute_reply":"2021-08-08T07:49:39.870652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('working/tmp')","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.874786Z","iopub.execute_input":"2021-08-08T07:49:39.875236Z","iopub.status.idle":"2021-08-08T07:49:39.881973Z","shell.execute_reply.started":"2021-08-08T07:49:39.875189Z","shell.execute_reply":"2021-08-08T07:49:39.880988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Move the images to relevant split folder.\nfor i in tqdm(range(len(df))):\n    row = df.loc[i]\n    if row.split == 'train':\n        copyfile(row.path, f'working/tmp/covid/images/train/{row.id}.jpg')\n    else:\n        copyfile(row.path, f'working/tmp/covid/images/valid/{row.id}.jpg')","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:49:39.883623Z","iopub.execute_input":"2021-08-08T07:49:39.884284Z","iopub.status.idle":"2021-08-08T07:50:29.707186Z","shell.execute_reply.started":"2021-08-08T07:49:39.884242Z","shell.execute_reply":"2021-08-08T07:50:29.705805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"yaml file will be used for training","metadata":{}},{"cell_type":"code","source":"# Create .yaml file \nimport yaml\n\ndata_yaml = dict(\n    train = '../covid/images/train',\n    val = '../covid/images/valid',\n    nc = 1,\n    names = ['opacity']\n)\n\n# Note that I am creating the file in the yolov5/data/ directory.\nwith open('working/tmp/yolov5/data/data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=True)\n    \n%cat working/tmp/yolov5/data/data.yaml","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:29.708703Z","iopub.execute_input":"2021-08-08T07:50:29.709064Z","iopub.status.idle":"2021-08-08T07:50:30.35827Z","shell.execute_reply.started":"2021-08-08T07:50:29.709028Z","shell.execute_reply":"2021-08-08T07:50:30.357275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the raw bounding box by parsing the row value of the label column.\n# Ref: https://www.kaggle.com/yujiariyasu/plot-3positive-classes\ndef get_bbox(row):\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row.label.split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l))\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []  \n            \n    return bboxes\n\n# Scale the bounding boxes according to the size of the resized image. \ndef scale_bbox(row, bboxes):\n    # Get scaling factor\n    scale_x = IMG_SIZE/row.dim1\n    scale_y = IMG_SIZE/row.dim0\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        x = int(np.round(bbox[0]*scale_x, 4))\n        y = int(np.round(bbox[1]*scale_y, 4))\n        x1 = int(np.round(bbox[2]*(scale_x), 4))\n        y1= int(np.round(bbox[3]*scale_y, 4))\n\n        scaled_bboxes.append([x, y, x1, y1]) # xmin, ymin, xmax, ymax\n        \n    return scaled_bboxes\n\n# Convert the bounding boxes in YOLO format.\ndef get_yolo_format_bbox(img_w, img_h, bboxes):\n    yolo_boxes = []\n    for bbox in bboxes:\n        w = bbox[2] - bbox[0] # xmax - xmin\n        h = bbox[3] - bbox[1] # ymax - ymin\n        xc = bbox[0] + int(np.round(w/2)) # xmin + width/2\n        yc = bbox[1] + int(np.round(h/2)) # ymin + height/2\n        \n        yolo_boxes.append([xc/img_w, yc/img_h, w/img_w, h/img_h]) # x_center y_center width height\n    \n    return yolo_boxes","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:30.360533Z","iopub.execute_input":"2021-08-08T07:50:30.361153Z","iopub.status.idle":"2021-08-08T07:50:30.37294Z","shell.execute_reply.started":"2021-08-08T07:50:30.361081Z","shell.execute_reply":"2021-08-08T07:50:30.372067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare the txt files for bounding box\nfor i in tqdm(range(len(df))):\n    row = df.loc[i]\n    # Get image id\n    img_id = row.id\n    # Get split\n    split = row.split\n    # Get image-level label\n    label = row.image_level\n    \n    if row.split=='train':\n        file_name = f'working/tmp/covid/labels/train/{row.id}.txt'\n    else:\n        file_name = f'working/tmp/covid/labels/valid/{row.id}.txt'\n        \n    \n    if label=='opacity':\n        # Get bboxes\n        bboxes = get_bbox(row)\n        # Scale bounding boxes\n        scale_bboxes = scale_bbox(row, bboxes)\n        # Format for YOLOv5\n        yolo_bboxes = get_yolo_format_bbox(IMG_SIZE, IMG_SIZE, scale_bboxes)\n        \n        with open(file_name, 'w') as f:\n            for bbox in yolo_bboxes:\n                bbox = [0]+bbox\n                bbox = [str(i) for i in bbox]\n                bbox = ' '.join(bbox)\n                f.write(bbox)\n                f.write('\\n')","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:30.37468Z","iopub.execute_input":"2021-08-08T07:50:30.375306Z","iopub.status.idle":"2021-08-08T07:50:33.070903Z","shell.execute_reply.started":"2021-08-08T07:50:30.375268Z","shell.execute_reply":"2021-08-08T07:50:33.069741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd working/tmp/yolov5/\n%cd /kaggle/working/tmp/yolov5/","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:25.499979Z","iopub.execute_input":"2021-08-09T14:05:25.500677Z","iopub.status.idle":"2021-08-09T14:05:25.511284Z","shell.execute_reply.started":"2021-08-09T14:05:25.500594Z","shell.execute_reply":"2021-08-09T14:05:25.510295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/working/tmp/yolov5/data/hyps')","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.080323Z","iopub.execute_input":"2021-08-08T07:50:33.080702Z","iopub.status.idle":"2021-08-08T07:50:33.095898Z","shell.execute_reply.started":"2021-08-08T07:50:33.080666Z","shell.execute_reply":"2021-08-08T07:50:33.095097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import yaml\n\nwith open('/kaggle/working/tmp/yolov5/data/hyps/hyp.scratch.yaml') as file:\n    # The FullLoader parameter handles the conversion from YAML\n    # scalar values to Python the dictionary format\n    fruits_list = yaml.load(file, Loader=yaml.FullLoader)\n\n    print(fruits_list)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.099009Z","iopub.execute_input":"2021-08-08T07:50:33.099291Z","iopub.status.idle":"2021-08-08T07:50:33.110696Z","shell.execute_reply.started":"2021-08-08T07:50:33.099265Z","shell.execute_reply":"2021-08-08T07:50:33.109793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fruits_list[\"lrf\"]= 0.032\nfruits_list[\"box\"]= 0.1\nfruits_list[\"cls\"]= 1.0\nfruits_list[\"cls_pw\"]= 0.5\nfruits_list[\"obj\"]= 2.0\nfruits_list[\"obj_pw\"]= 0.5\nfruits_list[\"anchors\"]= 0\nfruits_list[\"translate\"]= 0.2\nfruits_list[\"scale\"]= 0.6\nfruits_list[\"flipud\"]= 0.2\nfruits_list[\"fliplr\"]= 0.5","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.114512Z","iopub.execute_input":"2021-08-08T07:50:33.114776Z","iopub.status.idle":"2021-08-08T07:50:33.121634Z","shell.execute_reply.started":"2021-08-08T07:50:33.114751Z","shell.execute_reply":"2021-08-08T07:50:33.120498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dont change mosaic.\n# apply rotation\n# apply fliplr","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.123181Z","iopub.execute_input":"2021-08-08T07:50:33.123684Z","iopub.status.idle":"2021-08-08T07:50:33.129333Z","shell.execute_reply.started":"2021-08-08T07:50:33.123588Z","shell.execute_reply":"2021-08-08T07:50:33.128564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/working/tmp/yolov5/data/hyps/hyp.scratch.yaml\", 'w') as file:\n    documents = yaml.dump(fruits_list, file)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.130756Z","iopub.execute_input":"2021-08-08T07:50:33.131112Z","iopub.status.idle":"2021-08-08T07:50:33.140812Z","shell.execute_reply.started":"2021-08-08T07:50:33.131077Z","shell.execute_reply":"2021-08-08T07:50:33.139904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/working/tmp/yolov5/data/hyps/hyp.scratch.yaml') as file:\n    # The FullLoader parameter handles the conversion from YAML\n    # scalar values to Python the dictionary format\n    fruits_list = yaml.load(file, Loader=yaml.FullLoader)\n\n    print(fruits_list)","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.14417Z","iopub.execute_input":"2021-08-08T07:50:33.144465Z","iopub.status.idle":"2021-08-08T07:50:33.155896Z","shell.execute_reply.started":"2021-08-08T07:50:33.14443Z","shell.execute_reply":"2021-08-08T07:50:33.154727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## training model and test images prediction","metadata":{}},{"cell_type":"code","source":"!python train.py    --img {IMG_SIZE} \\\n                    --batch {BATCH_SIZE} \\\n                    --epochs {150} \\\n                    --data data.yaml \\\n                    --weights yolov5l.pt \\\n                    --cfg models/yolov5l.yaml\\\n                    --save_period 1 \\\n                    --project kaggle-siim-covid-yolov5l-clas1-mod8\n\n# here you can choose which model you want. Till now i have observed yolov5x gives best results","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:50:33.157425Z","iopub.execute_input":"2021-08-08T07:50:33.157827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"/kaggle/working/tmp/yolov5/runs\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python train.py    --img {IMG_SIZE} \\\n                    --batch {BATCH_SIZE} \\\n                    --epochs {120} \\\n                    --data data.yaml \\\n                    --weights /kaggle/working/tmp/yolov5/artifacts/run_2yubez04_model:v29/best.pt \\\n                    --save_period 1 \\\n                    --project kaggle-siim-covid-yolov5l-clas1-mod7","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:14:04.696085Z","iopub.execute_input":"2021-08-08T07:14:04.696448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now your model will be stored as an artifact on wandb account and also in yolov5 directory here. You can find command to predict results over testing data","metadata":{}},{"cell_type":"markdown","source":"Loading pretrained model artifacts from wandb. You can ignore this if you have just trained model. Just change save path","metadata":{}},{"cell_type":"code","source":"run = wandb.init()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:25.514694Z","iopub.execute_input":"2021-08-09T14:05:25.515014Z","iopub.status.idle":"2021-08-09T14:05:31.992037Z","shell.execute_reply.started":"2021-08-09T14:05:25.514986Z","shell.execute_reply":"2021-08-09T14:05:31.991079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# artifact = run.use_artifact(\"39ajinkya/kaggle-siim-covid-yolov5l-t3-clas1/run_1qerm3x5_model:v24\")\n# artifact = run.use_artifact(\"39ajinkya/kaggle-siim-covid-yolov5l-clas1-mod6/run_2yubez04_model:v29\")\nartifact = run.use_artifact(\"39ajinkya/kaggle-siim-covid-yolov5l-clas1-mod8/run_3l287lxi_model:v108\")\n\nartifact_dir = artifact.download()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:31.993760Z","iopub.execute_input":"2021-08-09T14:05:31.994145Z","iopub.status.idle":"2021-08-09T14:05:39.317386Z","shell.execute_reply.started":"2021-08-09T14:05:31.994105Z","shell.execute_reply":"2021-08-09T14:05:39.316563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# artifact = run.use_artifact(\"39ajinkya/kaggle-siim-covid-yolov5x-t1-clas1/run_lkm1qq0s_model:v19\")\n# artifact_dir = artifact.download()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:39.318853Z","iopub.execute_input":"2021-08-09T14:05:39.319258Z","iopub.status.idle":"2021-08-09T14:05:39.325815Z","shell.execute_reply.started":"2021-08-09T14:05:39.319217Z","shell.execute_reply":"2021-08-09T14:05:39.324964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# run_1qerm3x5_model:v24","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:39.327229Z","iopub.execute_input":"2021-08-09T14:05:39.327967Z","iopub.status.idle":"2021-08-09T14:05:39.342594Z","shell.execute_reply.started":"2021-08-09T14:05:39.327924Z","shell.execute_reply":"2021-08-09T14:05:39.341430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"run.join()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:39.347306Z","iopub.execute_input":"2021-08-09T14:05:39.349922Z","iopub.status.idle":"2021-08-09T14:05:42.792114Z","shell.execute_reply.started":"2021-08-09T14:05:39.349890Z","shell.execute_reply":"2021-08-09T14:05:42.791223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"/kaggle/working/tmp/yolov5/artifacts\")","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:42.793634Z","iopub.execute_input":"2021-08-09T14:05:42.794046Z","iopub.status.idle":"2021-08-09T14:05:42.800435Z","shell.execute_reply.started":"2021-08-09T14:05:42.794002Z","shell.execute_reply":"2021-08-09T14:05:42.799418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Put proper path in cell below and run detect.py to generate results**.","metadata":{}},{"cell_type":"code","source":"# MODEL_PATH = \"artifacts/run_2xb4vetk_model:v29/best.pt\"\n# MODEL_PATH = \"artifacts/run_1qerm3x5_model:v24/last.pt\"\nMODEL_PATH = \"artifacts/run_2yubez04_model:v29/best.pt\"","metadata":{"execution":{"iopub.status.busy":"2021-08-08T05:12:03.764515Z","iopub.execute_input":"2021-08-08T05:12:03.764878Z","iopub.status.idle":"2021-08-08T05:12:03.770531Z","shell.execute_reply.started":"2021-08-08T05:12:03.764846Z","shell.execute_reply":"2021-08-08T05:12:03.769755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MODEL_PATH = 'kaggle-siim-covid-yolov5l-t3-clas2/exp/weights/best.pt'\nTEST_PATH = '../../../input/siim-covid19-resized-to-512px-png/test/'","metadata":{"execution":{"iopub.status.busy":"2021-07-04T05:40:43.39529Z","iopub.execute_input":"2021-07-04T05:40:43.395693Z","iopub.status.idle":"2021-07-04T05:40:43.401723Z","shell.execute_reply.started":"2021-07-04T05:40:43.395651Z","shell.execute_reply":"2021-07-04T05:40:43.400849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights {MODEL_PATH} \\\n                  --source {TEST_PATH} \\\n                  --img {IMG_SIZE} \\\n                  --conf 0.3 \\\n                  --iou-thres 0.5 \\\n                  --max-det 3 \\\n                  --save-txt \\\n                  --save-conf","metadata":{"execution":{"iopub.status.busy":"2021-07-04T05:41:03.555981Z","iopub.execute_input":"2021-07-04T05:41:03.556329Z","iopub.status.idle":"2021-07-04T05:42:12.414517Z","shell.execute_reply.started":"2021-07-04T05:41:03.556299Z","shell.execute_reply":"2021-07-04T05:42:12.413581Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a = wandb.restore('39ajinkya/kaggle-siim-covid-yolov5x-t1-clas1/lkm1qq0s')\n# # 39ajinkya/kaggle-siim-covid-yolov5x-t1-clas1/lkm1qq0s","metadata":{"execution":{"iopub.status.busy":"2021-06-24T16:04:23.150895Z","iopub.execute_input":"2021-06-24T16:04:23.151337Z","iopub.status.idle":"2021-06-24T16:04:23.180526Z","shell.execute_reply.started":"2021-06-24T16:04:23.151295Z","shell.execute_reply":"2021-06-24T16:04:23.178353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python train.py --resume wandb-artifact://39ajinkya/kaggle-siim-covid-yolov5l-clas1-mod6/2yubez04 \\\n                 --epochs {150}\\\n# #!python train.py --resume MODEL_PATH                 ","metadata":{"execution":{"iopub.status.busy":"2021-08-08T07:09:40.306435Z","iopub.execute_input":"2021-08-08T07:09:40.306761Z","iopub.status.idle":"2021-08-08T07:09:57.768676Z","shell.execute_reply.started":"2021-08-08T07:09:40.306731Z","shell.execute_reply":"2021-08-08T07:09:57.767709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"/kaggle/working/tmp/yolov5/artifacts\")","metadata":{"execution":{"iopub.status.busy":"2021-07-31T13:52:52.555499Z","iopub.execute_input":"2021-07-31T13:52:52.555905Z","iopub.status.idle":"2021-07-31T13:52:52.563415Z","shell.execute_reply.started":"2021-07-31T13:52:52.555847Z","shell.execute_reply":"2021-07-31T13:52:52.562199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# os.listdir('/kaggle/tmp/yolov5/runs/detect/exp/labels')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T11:59:09.57632Z","iopub.execute_input":"2021-06-23T11:59:09.576675Z","iopub.status.idle":"2021-06-23T11:59:09.585789Z","shell.execute_reply.started":"2021-06-23T11:59:09.576639Z","shell.execute_reply":"2021-06-23T11:59:09.584295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PRED_PATH = 'runs/detect/exp/labels'  # it can be exp/exp2/exp3 depending on your count of running detect.py. First run will save results in exp\n!ls {PRED_PATH}","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:42:22.617996Z","iopub.execute_input":"2021-06-23T15:42:22.618384Z","iopub.status.idle":"2021-06-23T15:42:23.256272Z","shell.execute_reply.started":"2021-06-23T15:42:22.618347Z","shell.execute_reply":"2021-06-23T15:42:23.2553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize predicted coordinates.\n%cat runs/detect/exp3/labels/ba91d37ee459.txt","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:42:33.85355Z","iopub.execute_input":"2021-06-23T15:42:33.853914Z","iopub.status.idle":"2021-06-23T15:42:34.489431Z","shell.execute_reply.started":"2021-06-23T15:42:33.853881Z","shell.execute_reply":"2021-06-23T15:42:34.488496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_files = os.listdir(PRED_PATH)\nprint('Number of test images predicted as opaque: ', len(prediction_files))","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:42:42.542727Z","iopub.execute_input":"2021-06-23T15:42:42.543087Z","iopub.status.idle":"2021-06-23T15:42:42.54904Z","shell.execute_reply.started":"2021-06-23T15:42:42.543046Z","shell.execute_reply":"2021-06-23T15:42:42.548178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Store results in submission.csv file  ","metadata":{}},{"cell_type":"code","source":"# The submisison requires xmin, ymin, xmax, ymax format. \n# YOLOv5 returns x_center, y_center, width, height\ndef correct_bbox_format(bboxes):\n    correct_bboxes = []\n    for b in bboxes:\n        xc, yc = int(np.round(b[0]*IMG_SIZE)), int(np.round(b[1]*IMG_SIZE))\n        w, h = int(np.round(b[2]*IMG_SIZE)), int(np.round(b[3]*IMG_SIZE))\n\n        xmin = xc - int(np.round(w/2))\n        xmax = xc + int(np.round(w/2))\n        ymin = yc - int(np.round(h/2))\n        ymax = yc + int(np.round(h/2))\n        \n        correct_bboxes.append([xmin, xmax, ymin, ymax])\n        \n    return correct_bboxes\n\n# Read the txt file generated by YOLOv5 during inference and extract \n# confidence and bounding box coordinates.\ndef get_conf_bboxes(file_path):\n    confidence = []\n    bboxes = []\n    with open(file_path, 'r') as file:\n        for line in file:\n            preds = line.strip('\\n').split(' ')\n            preds = list(map(float, preds))\n            confidence.append(preds[-1])\n            bboxes.append(preds[1:-1])\n    return confidence, bboxes","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:42:45.485146Z","iopub.execute_input":"2021-06-23T15:42:45.485484Z","iopub.status.idle":"2021-06-23T15:42:45.494345Z","shell.execute_reply.started":"2021-06-23T15:42:45.485456Z","shell.execute_reply":"2021-06-23T15:42:45.493511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the submisison file\nsub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nsub_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:42:48.556204Z","iopub.execute_input":"2021-06-23T15:42:48.556524Z","iopub.status.idle":"2021-06-23T15:42:48.576854Z","shell.execute_reply.started":"2021-06-23T15:42:48.556495Z","shell.execute_reply":"2021-06-23T15:42:48.575884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prediction loop for submission\npredictions = []\n\nfor i in tqdm(range(len(sub_df))):\n    row = sub_df.loc[i]\n    id_name = row.id.split('_')[0]\n    id_level = row.id.split('_')[-1]\n    \n    if id_level == 'study':\n        # do study-level classification\n        predictions.append(\"Negative 1 0 0 1 1\") # dummy prediction\n        \n    elif id_level == 'image':\n        # we can do image-level classification here.\n        # also we can rely on the object detector's classification head.\n        # for this example submisison we will use YOLO's classification head. \n        # since we already ran the inference we know which test images belong to opacity.\n        if f'{id_name}.txt' in prediction_files:\n            # opacity label\n            confidence, bboxes = get_conf_bboxes(f'{PRED_PATH}/{id_name}.txt')\n            bboxes = correct_bbox_format(bboxes)\n            pred_string = ''\n            for j, conf in enumerate(confidence):\n                pred_string += f'opacity {conf} ' + ' '.join(map(str, bboxes[j])) + ' '\n            predictions.append(pred_string[:-1]) \n        else:\n            predictions.append(\"None 1 0 0 1 1\")","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:42:54.048202Z","iopub.execute_input":"2021-06-23T15:42:54.048521Z","iopub.status.idle":"2021-06-23T15:42:54.491177Z","shell.execute_reply.started":"2021-06-23T15:42:54.048491Z","shell.execute_reply":"2021-06-23T15:42:54.490171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['PredictionString'] = predictions\nsub_df.to_csv('/kaggle/working/submission.csv', index=False)\nsub_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:43:03.759536Z","iopub.execute_input":"2021-06-23T15:43:03.759873Z","iopub.status.idle":"2021-06-23T15:43:04.110805Z","shell.execute_reply.started":"2021-06-23T15:43:03.759841Z","shell.execute_reply":"2021-06-23T15:43:04.109982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.loc[sub_df['PredictionString'] == \"None 1 0 0 1 1\"]","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:23:13.229827Z","iopub.execute_input":"2021-06-23T13:23:13.230283Z","iopub.status.idle":"2021-06-23T13:23:13.304521Z","shell.execute_reply.started":"2021-06-23T13:23:13.230246Z","shell.execute_reply":"2021-06-23T13:23:13.302739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/working')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T12:48:15.82343Z","iopub.execute_input":"2021-06-23T12:48:15.823801Z","iopub.status.idle":"2021-06-23T12:48:15.828991Z","shell.execute_reply.started":"2021-06-23T12:48:15.82375Z","shell.execute_reply":"2021-06-23T12:48:15.82814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"creating a coco format dataset.","metadata":{}},{"cell_type":"code","source":"# #creating a coco format dataset.\n# df.iloc[0].path","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:02:16.034664Z","iopub.execute_input":"2021-06-18T13:02:16.03501Z","iopub.status.idle":"2021-06-18T13:02:16.041594Z","shell.execute_reply.started":"2021-06-18T13:02:16.034979Z","shell.execute_reply":"2021-06-18T13:02:16.040506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annotation = dict()\n\n# for i in tqdm(range(len(df))):\n#     row = df.loc[i]\n#     # Get image id\n#     img_id = row.id\n#     # Get split\n#     split = row.split\n#     # Get image-level label\n#     label = row.image_level\n    \n#     file_name = row.path\n        \n    \n#     if label=='opacity':\n#         # Get bboxes\n#         bboxes = get_bbox(row)\n#         # Scale bounding boxes\n#         scale_bboxes = scale_bbox(row, bboxes)\n#         # Format for YOLOv5\n#         yolo_bboxes = get_yolo_format_bbox(IMG_SIZE, IMG_SIZE, scale_bboxes)\n        \n#         l = []\n#         for bbox in yolo_bboxes:\n#             l.append({'bbox':bbox,'label':'opacity'})\n#         annotation[file_name] = l","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:23:25.924899Z","iopub.execute_input":"2021-06-18T13:23:25.925238Z","iopub.status.idle":"2021-06-18T13:23:27.853252Z","shell.execute_reply.started":"2021-06-18T13:23:25.925209Z","shell.execute_reply":"2021-06-18T13:23:27.852401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# annotation","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:23:29.554938Z","iopub.execute_input":"2021-06-18T13:23:29.55525Z","iopub.status.idle":"2021-06-18T13:23:29.793865Z","shell.execute_reply.started":"2021-06-18T13:23:29.555218Z","shell.execute_reply":"2021-06-18T13:23:29.793033Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del dataset","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:10:27.635044Z","iopub.execute_input":"2021-06-18T13:10:27.635371Z","iopub.status.idle":"2021-06-18T13:10:27.63875Z","shell.execute_reply.started":"2021-06-18T13:10:27.63534Z","shell.execute_reply":"2021-06-18T13:10:27.637847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:17:22.02986Z","iopub.execute_input":"2021-06-18T13:17:22.030192Z","iopub.status.idle":"2021-06-18T13:17:22.035232Z","shell.execute_reply.started":"2021-06-18T13:17:22.030161Z","shell.execute_reply":"2021-06-18T13:17:22.034351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_path = 'kaggle/input/siim-covid19-resized-to-256px-jpg/train/*'\n# glob.glob(image_path)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:19:32.579709Z","iopub.execute_input":"2021-06-18T13:19:32.580021Z","iopub.status.idle":"2021-06-18T13:19:32.624363Z","shell.execute_reply.started":"2021-06-18T13:19:32.579991Z","shell.execute_reply":"2021-06-18T13:19:32.623434Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import glob\n# import fiftyone as fo\n\n# image_path = 'kaggle/input/siim-covid19-resized-to-256px-jpg/train/*'\n\n# # Ex: your custom label format\n\n# # Create dataset\n# dataset = fo.Dataset(name=\"siim-covid-19-6\")\n\n# # Persist the dataset on disk in order to \n# # be able to load it in one line in the future\n# dataset.persistent = True\n\n# # Add your samples to the dataset\n# for filepath in annotation:\n#     sample = fo.Sample(filepath=filepath)\n#     sample.tags.append('images')\n#     # Convert detections to FiftyOne format\n#     detections = []\n#     for obj in annotation[filepath]:\n#         label = obj[\"label\"]\n\n#         # Bounding box coordinates should be relative values\n#         # in [0, 1] in the following format:\n#         # [top-left-x, top-left-y, width, height]\n#         bounding_box = obj[\"bbox\"]\n        \n#         detections.append(\n#             fo.Detection(label=label, bounding_box=bounding_box)\n#         )\n\n#     # Store detections in a field name of your choice\n#     sample[\"train\"] = fo.Detections(detections=detections)\n\n#     dataset.add_sample(sample)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:25:35.59564Z","iopub.execute_input":"2021-06-18T13:25:35.595953Z","iopub.status.idle":"2021-06-18T13:25:45.163808Z","shell.execute_reply.started":"2021-06-18T13:25:35.595917Z","shell.execute_reply":"2021-06-18T13:25:45.162775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view = dataset.match_tags('images')\n# for sample in view:\n#     print(sample)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:25:49.03443Z","iopub.execute_input":"2021-06-18T13:25:49.034752Z","iopub.status.idle":"2021-06-18T13:26:10.008087Z","shell.execute_reply.started":"2021-06-18T13:25:49.034721Z","shell.execute_reply":"2021-06-18T13:26:10.007317Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# export_dir = \"/path/for/coco-detection-dataset-1\"\n# label_field = \"ground_truth\"  # for example\n\n# # Export the dataset\n# dataset.export(\n#     export_dir=export_dir,\n#     dataset_type=fo.types.COCODetectionDataset,\n#     label_field=label_field,\n# )","metadata":{"execution":{"iopub.status.busy":"2021-06-18T13:06:03.990294Z","iopub.execute_input":"2021-06-18T13:06:03.99067Z","iopub.status.idle":"2021-06-18T13:06:04.001856Z","shell.execute_reply.started":"2021-06-18T13:06:03.990638Z","shell.execute_reply":"2021-06-18T13:06:04.000818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install fiftyone","metadata":{"execution":{"iopub.status.busy":"2021-06-18T12:50:53.42484Z","iopub.execute_input":"2021-06-18T12:50:53.42518Z","iopub.status.idle":"2021-06-18T12:51:19.961086Z","shell.execute_reply.started":"2021-06-18T12:50:53.425148Z","shell.execute_reply":"2021-06-18T12:51:19.960121Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}