{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\n\n\n\n\n# Necessary/extra dependencies. \nimport os\nimport gc\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom shutil import copyfile\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n#! conda install -c conda-forge gdcm -y\n#! conda install pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg\n#! conda install pillow\n#customize iPython writefile so we can write variables\nfrom IPython.core.magic import register_line_cell_magic\n\n\nimport pylab\n#import pillow\n#import gdcm\n#pydicom\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom fastai.imports import *\n#from fastai.medical.imaging import *\nfrom PIL import Image\n\n@register_line_cell_magic\ndef writetemplate(line, cell):\n    with open(line, 'w') as f:\n        f.write(cell.format(**globals()))\n        \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-09T14:31:35.337172Z","iopub.execute_input":"2021-08-09T14:31:35.337603Z","iopub.status.idle":"2021-08-09T14:31:36.508837Z","shell.execute_reply.started":"2021-08-09T14:31:35.337512Z","shell.execute_reply":"2021-08-09T14:31:36.507975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clone  Yolo  from GIT","metadata":{}},{"cell_type":"code","source":"# Download YOLOv5\n!git clone https://github.com/ultralytics/yolov5  # clone repo\n%cd yolov5\n# Install dependencies\n%pip install -qr requirements.txt  # install dependencies\n\n%cd ../\nimport torch\nprint(f\"Setup complete. Using torch {torch.__version__} ({torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else 'CPU'})\")","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:31:36.510260Z","iopub.execute_input":"2021-08-09T14:31:36.510595Z","iopub.status.idle":"2021-08-09T14:31:49.671711Z","shell.execute_reply.started":"2021-08-09T14:31:36.510559Z","shell.execute_reply":"2021-08-09T14:31:49.670678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install W&B \n!pip install -q --upgrade wandb\n# Login \nimport wandb\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:31:49.675266Z","iopub.execute_input":"2021-08-09T14:31:49.675541Z","iopub.status.idle":"2021-08-09T14:32:13.644288Z","shell.execute_reply.started":"2021-08-09T14:31:49.675514Z","shell.execute_reply":"2021-08-09T14:32:13.643430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read Files","metadata":{}},{"cell_type":"code","source":"\ndf=pd.read_csv('/kaggle/input/df-train/df_train.csv')\n\nTRAIN_PATH='/kaggle/input/siim-covid19-resized-to-256px-jpg/train/'\n\n# TRAIN_PATH=  '/kaggle/working/siim-covid19-resized-to-256px-jpg/train/'\n# Add absolute path\ndf['path'] = df.apply(lambda row: TRAIN_PATH+row.id+'.jpg', axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:13.647621Z","iopub.execute_input":"2021-08-09T14:32:13.647889Z","iopub.status.idle":"2021-08-09T14:32:13.833636Z","shell.execute_reply.started":"2021-08-09T14:32:13.647861Z","shell.execute_reply":"2021-08-09T14:32:13.832731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['path'][0]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:13.836802Z","iopub.execute_input":"2021-08-09T14:32:13.837092Z","iopub.status.idle":"2021-08-09T14:32:13.844750Z","shell.execute_reply.started":"2021-08-09T14:32:13.837062Z","shell.execute_reply":"2021-08-09T14:32:13.843659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = df[['Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']].values\nlabels = np.argmax(labels, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:13.846430Z","iopub.execute_input":"2021-08-09T14:32:13.847128Z","iopub.status.idle":"2021-08-09T14:32:13.856435Z","shell.execute_reply.started":"2021-08-09T14:32:13.847084Z","shell.execute_reply":"2021-08-09T14:32:13.855365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['label_y','label_int']]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:13.858085Z","iopub.execute_input":"2021-08-09T14:32:13.858470Z","iopub.status.idle":"2021-08-09T14:32:13.876923Z","shell.execute_reply.started":"2021-08-09T14:32:13.858417Z","shell.execute_reply":"2021-08-09T14:32:13.876159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# dim 0 -->h\n# dim 1 -->w","metadata":{}},{"cell_type":"code","source":"df['x_min'] = df.apply(lambda row: (row.x_min)/row.w, axis =1)\ndf['y_min'] = df.apply(lambda row: (row.y_min)/row.h, axis =1)\n\ndf['x_max'] = df.apply(lambda row: (row.x_max)/row.w, axis =1)\ndf['y_max'] = df.apply(lambda row: (row.y_max)/row.h, axis =1)\n\ndf['x_mid'] = df.apply(lambda row: (row.x_max+row.x_min)/2, axis =1)\ndf['y_mid'] = df.apply(lambda row: (row.y_max+row.y_min)/2, axis =1)\n\ndf['w'] = df.apply(lambda row: (row.x_max-row.x_min), axis =1)\ndf['h'] = df.apply(lambda row: (row.y_max-row.y_min), axis =1)\n\ndf['area'] = df['w']*df['h']\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:13.879528Z","iopub.execute_input":"2021-08-09T14:32:13.879810Z","iopub.status.idle":"2021-08-09T14:32:15.069903Z","shell.execute_reply.started":"2021-08-09T14:32:13.879785Z","shell.execute_reply":"2021-08-09T14:32:15.069071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df['class']\ndf['image_level'] = df.apply(lambda row: row.label.split(' ')[0], axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.071700Z","iopub.execute_input":"2021-08-09T14:32:15.071977Z","iopub.status.idle":"2021-08-09T14:32:15.178360Z","shell.execute_reply.started":"2021-08-09T14:32:15.071948Z","shell.execute_reply":"2021-08-09T14:32:15.177512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Create train and validation split.\ntrain_df, valid_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df.image_level.values)\n\ntrain_df.loc[:, 'split'] = 'train'\nvalid_df.loc[:, 'split'] = 'valid'\n\ndf = pd.concat([train_df, valid_df]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.179748Z","iopub.execute_input":"2021-08-09T14:32:15.180091Z","iopub.status.idle":"2021-08-09T14:32:15.216411Z","shell.execute_reply.started":"2021-08-09T14:32:15.180054Z","shell.execute_reply":"2021-08-09T14:32:15.215536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#TRAIN_PATH = 'input/siim-covid19-resized-to-256px-jpg/train/'\nIMG_SIZE = 256\nBATCH_SIZE = 16\nEPOCHS = 10\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.217797Z","iopub.execute_input":"2021-08-09T14:32:15.218139Z","iopub.status.idle":"2021-08-09T14:32:15.222752Z","shell.execute_reply.started":"2021-08-09T14:32:15.218102Z","shell.execute_reply":"2021-08-09T14:32:15.221545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create Images DataSets","metadata":{}},{"cell_type":"code","source":"#import pylibjpeg \nfrom fastai.imports import *\nfrom fastai.medical.imaging import *\ndef loadfilename(filename,voi_lut = True, fix_monochrome = True):\n    \n    \n    information={}\n\n    img = pydicom.read_file(filename)   \n\n\n    information['PatientID'] = img.PatientID\n\n    information['PatientName'] = img.PatientName\n\n    information['PatientSex'] = img.PatientSex\n\n    information['StudyID'] = img.StudyID\n\n    information['StudyDate'] = img.StudyDate\n\n    information['StudyTime'] = img.StudyTime\n    \n    if voi_lut:\n        img_data = apply_voi_lut(img.pixel_array, img)\n        \n    else:\n        img_data=img.pixel_array\n        \n    if fix_monochrome and img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_data = np.amax(img_data) - img_data  \n\n    #print(np.max(img_data))\n    #print(np.min(img_data))\n\n    img_data=img_data-np.min(img_data)\n    img_data=img_data/np.max(img_data)\n    img_data=(img_data * 255).astype(np.uint8)\n\n    # return information,img_data\n    return img_data","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.224756Z","iopub.execute_input":"2021-08-09T14:32:15.225209Z","iopub.status.idle":"2021-08-09T14:32:15.909253Z","shell.execute_reply.started":"2021-08-09T14:32:15.225169Z","shell.execute_reply":"2021-08-09T14:32:15.906572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# os.makedirs('siim-covid19-resized-to-256px-jpg/train',exist_ok=True)\n\ndef create_dataset():\n    for i in  tqdm(range(len(df))):\n        row=df.loc[i]\n        img_name=row.id\n        study_id=(row.StudyInstanceUID)\n        dicom_path= (\"../input/siim-covid19-detection/train/{}\".format(study_id))\n\n        path_x=(os.path.join(dicom_path,os.listdir(dicom_path)[0]))\n        img_path=os.path.join(path_x,os.listdir(path_x)[0])\n        #print(img_path)\n        #info,img=loadfilename(img_path)\n        img=loadfilename(img_path)\n        img_s = cv2.resize(img, (IMG_SIZE,IMG_SIZE))\n        #print('siim-covid19-resized-to-256px-jpg/train/'+str(img_name)+\".jpg\")\n        cv2.imwrite('siim-covid19-resized-to-256px-jpg/train/'+str(img_name)+\".jpg\", img_s)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.912491Z","iopub.execute_input":"2021-08-09T14:32:15.913181Z","iopub.status.idle":"2021-08-09T14:32:15.923108Z","shell.execute_reply.started":"2021-08-09T14:32:15.913136Z","shell.execute_reply":"2021-08-09T14:32:15.921665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load meta.csv file\n# Original dimensions are required to scale the bounding box coordinates appropriately.\nmeta_df = pd.read_csv('/kaggle/input/siim-covid19-resized-to-256px-jpg/meta.csv')\n\ntrain_meta_df = meta_df.loc[meta_df.split == 'train']\ntrain_meta_df = train_meta_df.drop('split', axis=1)\ntrain_meta_df.columns = ['id', 'dim0', 'dim1']\n\ntrain_meta_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.924936Z","iopub.execute_input":"2021-08-09T14:32:15.925411Z","iopub.status.idle":"2021-08-09T14:32:15.973492Z","shell.execute_reply.started":"2021-08-09T14:32:15.925371Z","shell.execute_reply":"2021-08-09T14:32:15.971948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(train_meta_df, on='id',how=\"left\")\ndf.head(2)\n\ndf[['w','h','dim0','dim1']]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:15.975297Z","iopub.execute_input":"2021-08-09T14:32:15.975861Z","iopub.status.idle":"2021-08-09T14:32:16.028280Z","shell.execute_reply.started":"2021-08-09T14:32:15.975824Z","shell.execute_reply":"2021-08-09T14:32:16.027472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Size of dataset: {len(df)}, training images: {len(train_df)}. validation images: {len(valid_df)}')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:16.030167Z","iopub.execute_input":"2021-08-09T14:32:16.030944Z","iopub.status.idle":"2021-08-09T14:32:16.038310Z","shell.execute_reply.started":"2021-08-09T14:32:16.030878Z","shell.execute_reply":"2021-08-09T14:32:16.037153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('covid/images/train', exist_ok=True)\nos.makedirs('covid/images/valid', exist_ok=True)\n\nos.makedirs('covid/labels/train', exist_ok=True)\nos.makedirs('covid/labels/valid', exist_ok=True)\n\n! ls covid/images","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:16.039892Z","iopub.execute_input":"2021-08-09T14:32:16.040480Z","iopub.status.idle":"2021-08-09T14:32:16.863073Z","shell.execute_reply.started":"2021-08-09T14:32:16.040413Z","shell.execute_reply":"2021-08-09T14:32:16.862005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im_path='/kaggle/input/siim-covid19-resized-to-256px-jpg/train'\nim_path_list=(os.listdir(im_path))\nprint('b9175a64ad09.jpg' in im_path_list)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:16.867054Z","iopub.execute_input":"2021-08-09T14:32:16.870052Z","iopub.status.idle":"2021-08-09T14:32:17.059794Z","shell.execute_reply.started":"2021-08-09T14:32:16.869979Z","shell.execute_reply":"2021-08-09T14:32:17.058964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get Boxes","metadata":{}},{"cell_type":"code","source":"# Get the raw bounding box by parsing the row value of the label column.\n# Ref: https://www.kaggle.com/yujiariyasu/plot-3positive-classes\ndef get_bbox(row):\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row.label.split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l))\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []  \n            \n    return bboxes\n\n# Scale the bounding boxes according to the size of the resized image. \ndef scale_bbox(row, bboxes):\n    # Get scaling factor\n    scale_x = IMG_SIZE/row.dim1\n    scale_y = IMG_SIZE/row.dim0\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        x = int(np.round(bbox[0]*scale_x, 4))\n        y = int(np.round(bbox[1]*scale_y, 4))\n        x1 = int(np.round(bbox[2]*(scale_x), 4))\n        y1= int(np.round(bbox[3]*scale_y, 4))\n\n        scaled_bboxes.append([x, y, x1, y1]) # xmin, ymin, xmax, ymax\n        \n    return scaled_bboxes\n\n# Convert the bounding boxes in YOLO format.\ndef get_yolo_format_bbox(img_w, img_h, bboxes):\n    yolo_boxes = []\n    for bbox in bboxes:\n        w = bbox[2] - bbox[0] # xmax - xmin\n        h = bbox[3] - bbox[1] # ymax - ymin\n        xc = bbox[0] + int(np.round(w/2)) # xmin + width/2\n        yc = bbox[1] + int(np.round(h/2)) # ymin + height/2\n        \n        yolo_boxes.append([xc/img_w, yc/img_h, w/img_w, h/img_h]) # x_center y_center width height\n    \n    return yolo_boxes","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:17.062627Z","iopub.execute_input":"2021-08-09T14:32:17.063292Z","iopub.status.idle":"2021-08-09T14:32:17.078535Z","shell.execute_reply.started":"2021-08-09T14:32:17.063259Z","shell.execute_reply":"2021-08-09T14:32:17.077698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from shutil import copyfile\ndef create_img_dataset():\n    # Move the images to relevant split folder.\n    for i in tqdm(range(len(df))):\n        row = df.loc[i]\n        if row.split == 'train':\n            copyfile(row.path, f'covid/images/train/{row.id}.jpg')\n        else:\n            copyfile(row.path, f'covid/images/valid/{row.id}.jpg')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:17.080472Z","iopub.execute_input":"2021-08-09T14:32:17.080755Z","iopub.status.idle":"2021-08-09T14:32:17.090164Z","shell.execute_reply.started":"2021-08-09T14:32:17.080723Z","shell.execute_reply":"2021-08-09T14:32:17.088536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_img_dataset()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:17.092653Z","iopub.execute_input":"2021-08-09T14:32:17.092921Z","iopub.status.idle":"2021-08-09T14:32:52.543462Z","shell.execute_reply.started":"2021-08-09T14:32:17.092886Z","shell.execute_reply":"2021-08-09T14:32:52.541765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare the txt files for bounding box\nfor i in tqdm(range(len(df))):\n    row = df.loc[i]\n    # Get image id\n    img_id = row.id\n    # Get split\n    split = row.split\n    # Get image-level label\n    label = row.image_level\n    \n    if row.split=='train':\n        file_name = f'covid/labels/train/{row.id}.txt'\n    else:\n        file_name = f'covid/labels/valid/{row.id}.txt'\n        \n    \n    if label=='opacity':\n        # Get bboxes\n        bboxes = get_bbox(row)\n        # Scale bounding boxes\n        scale_bboxes = scale_bbox(row, bboxes)\n        # Format for YOLOv5\n        yolo_bboxes = get_yolo_format_bbox(IMG_SIZE, IMG_SIZE, scale_bboxes)\n        \n        with open(file_name, 'w') as f:\n            for bbox in yolo_bboxes:\n                bbox = [1]+bbox\n                bbox = [str(i) for i in bbox]\n                bbox = ' '.join(bbox)\n                f.write(bbox)\n                f.write('\\n')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:52.545041Z","iopub.execute_input":"2021-08-09T14:32:52.545414Z","iopub.status.idle":"2021-08-09T14:32:55.551856Z","shell.execute_reply.started":"2021-08-09T14:32:52.545377Z","shell.execute_reply":"2021-08-09T14:32:55.550865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dowload data set in zip folder..!!**","metadata":{}},{"cell_type":"code","source":"\n#! zip -r output.zip covid\n\n#! rm -rf covid\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:55.557153Z","iopub.execute_input":"2021-08-09T14:32:55.557413Z","iopub.status.idle":"2021-08-09T14:32:55.561167Z","shell.execute_reply.started":"2021-08-09T14:32:55.557387Z","shell.execute_reply":"2021-08-09T14:32:55.560313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create .yaml file \nimport yaml","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:55.564171Z","iopub.execute_input":"2021-08-09T14:32:55.564727Z","iopub.status.idle":"2021-08-09T14:32:55.572511Z","shell.execute_reply.started":"2021-08-09T14:32:55.564689Z","shell.execute_reply":"2021-08-09T14:32:55.571243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Yolo","metadata":{}},{"cell_type":"code","source":"%cd yolov5","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:55.573915Z","iopub.execute_input":"2021-08-09T14:32:55.574339Z","iopub.status.idle":"2021-08-09T14:32:55.583836Z","shell.execute_reply.started":"2021-08-09T14:32:55.574304Z","shell.execute_reply":"2021-08-09T14:32:55.582719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(os.listdir('/kaggle/working/yolov5/data'))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:55.585531Z","iopub.execute_input":"2021-08-09T14:32:55.586331Z","iopub.status.idle":"2021-08-09T14:32:55.595075Z","shell.execute_reply.started":"2021-08-09T14:32:55.586276Z","shell.execute_reply":"2021-08-09T14:32:55.594034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_yaml = dict(\n    train = '/kaggle/working/covid/images/train',\n    val = '/kaggle/working/covid/images/valid',\n    nc = 2,\n    names = ['none', 'opacity']\n)\n\n# Note that I am creating the file in the yolov5/data/ directory.\nwith open('/kaggle/working/yolov5/data/data.yaml', 'w') as outfile:\n    yaml.dump(data_yaml, outfile, default_flow_style=True)\n    \n    \n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:55.596984Z","iopub.execute_input":"2021-08-09T14:32:55.597658Z","iopub.status.idle":"2021-08-09T14:32:55.606642Z","shell.execute_reply.started":"2021-08-09T14:32:55.597615Z","shell.execute_reply":"2021-08-09T14:32:55.605433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n%cat /kaggle/working/yolov5/data/data.yaml","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:55.608693Z","iopub.execute_input":"2021-08-09T14:32:55.609128Z","iopub.status.idle":"2021-08-09T14:32:56.292547Z","shell.execute_reply.started":"2021-08-09T14:32:55.609050Z","shell.execute_reply":"2021-08-09T14:32:56.291509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('/kaggle/working/covid/images/valid'))\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:56.296018Z","iopub.execute_input":"2021-08-09T14:32:56.296317Z","iopub.status.idle":"2021-08-09T14:32:56.308275Z","shell.execute_reply.started":"2021-08-09T14:32:56.296285Z","shell.execute_reply":"2021-08-09T14:32:56.307061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir('/kaggle/working/covid/images/train'))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:56.310973Z","iopub.execute_input":"2021-08-09T14:32:56.311231Z","iopub.status.idle":"2021-08-09T14:32:56.326171Z","shell.execute_reply.started":"2021-08-09T14:32:56.311199Z","shell.execute_reply":"2021-08-09T14:32:56.325292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python train.py --img {IMG_SIZE} \\\n                 --batch {BATCH_SIZE} \\\n                 --epochs {EPOCHS} \\\n                 --data data.yaml \\\n                 --weights yolov5m.pt \\\n                 --project kaggle-siim-covid19 \\\n                 --cache","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:32:56.329231Z","iopub.execute_input":"2021-08-09T14:32:56.329544Z","iopub.status.idle":"2021-08-09T15:06:06.050001Z","shell.execute_reply.started":"2021-08-09T14:32:56.329515Z","shell.execute_reply":"2021-08-09T15:06:06.048937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (20,20))\nplt.axis('off')\nplt.imshow(plt.imread('/kaggle/working/yolov5/kaggle-siim-covid19/exp/labels.jpg'));","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:06.051715Z","iopub.execute_input":"2021-08-09T15:06:06.052077Z","iopub.status.idle":"2021-08-09T15:06:06.905165Z","shell.execute_reply.started":"2021-08-09T15:06:06.052035Z","shell.execute_reply":"2021-08-09T15:06:06.904279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls '/kaggle/working/yolov5/kaggle-siim-covid19/exp'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:06.906220Z","iopub.execute_input":"2021-08-09T15:06:06.906566Z","iopub.status.idle":"2021-08-09T15:06:07.592101Z","shell.execute_reply.started":"2021-08-09T15:06:06.906524Z","shell.execute_reply":"2021-08-09T15:06:07.591029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = '/kaggle/input/siim-covid19-resized-to-256px-jpg/test/'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.596651Z","iopub.execute_input":"2021-08-09T15:06:07.598701Z","iopub.status.idle":"2021-08-09T15:06:07.607021Z","shell.execute_reply.started":"2021-08-09T15:06:07.598656Z","shell.execute_reply":"2021-08-09T15:06:07.606084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_dir = 'kaggle-siim-covid19/exp/weights/best.pt'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.608630Z","iopub.execute_input":"2021-08-09T15:06:07.610707Z","iopub.status.idle":"2021-08-09T15:06:07.618037Z","shell.execute_reply.started":"2021-08-09T15:06:07.610666Z","shell.execute_reply":"2021-08-09T15:06:07.617199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd '/kaggle/working/yolov5'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.619603Z","iopub.execute_input":"2021-08-09T15:06:07.621663Z","iopub.status.idle":"2021-08-09T15:06:07.632869Z","shell.execute_reply.started":"2021-08-09T15:06:07.621626Z","shell.execute_reply":"2021-08-09T15:06:07.631877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # Run Detection","metadata":{}},{"cell_type":"code","source":"\n\n#!python detect.py --weights {weights_dir} \\\n#                  --source {TEST_PATH} \\\n#                  --img {IMG_SIZE} \\\n#                  --conf 0.28 \\\n#                  --iou-thres 0.5 \\\n#                  --max-det 3 \\\n#                  --save-txt \\\n#                  --save-conf \\\n #                 --exist-ok\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.635860Z","iopub.execute_input":"2021-08-09T15:06:07.637538Z","iopub.status.idle":"2021-08-09T15:06:07.643372Z","shell.execute_reply.started":"2021-08-09T15:06:07.637493Z","shell.execute_reply":"2021-08-09T15:06:07.642550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pred_label_list=os.listdir('runs/detect/exp/labels/')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.645474Z","iopub.execute_input":"2021-08-09T15:06:07.646081Z","iopub.status.idle":"2021-08-09T15:06:07.653394Z","shell.execute_reply.started":"2021-08-09T15:06:07.646041Z","shell.execute_reply":"2021-08-09T15:06:07.652546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(f'Number of opacity predicted by YOLOv5: {len(pred_label_list)}')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.654755Z","iopub.execute_input":"2021-08-09T15:06:07.655221Z","iopub.status.idle":"2021-08-09T15:06:07.661333Z","shell.execute_reply.started":"2021-08-09T15:06:07.655180Z","shell.execute_reply":"2021-08-09T15:06:07.660572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n#!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n#!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n#!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n#!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n#!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.662713Z","iopub.execute_input":"2021-08-09T15:06:07.664370Z","iopub.status.idle":"2021-08-09T15:06:07.671535Z","shell.execute_reply.started":"2021-08-09T15:06:07.664329Z","shell.execute_reply":"2021-08-09T15:06:07.670724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the submisison file\nsub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nprint(len(sub_df))\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.673512Z","iopub.execute_input":"2021-08-09T15:06:07.674037Z","iopub.status.idle":"2021-08-09T15:06:07.722638Z","shell.execute_reply.started":"2021-08-09T15:06:07.673998Z","shell.execute_reply":"2021-08-09T15:06:07.721823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df = sub_df.loc[sub_df.id.str.contains('_study')]\nlen(study_df)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.723910Z","iopub.execute_input":"2021-08-09T15:06:07.724421Z","iopub.status.idle":"2021-08-09T15:06:07.745389Z","shell.execute_reply.started":"2021-08-09T15:06:07.724382Z","shell.execute_reply":"2021-08-09T15:06:07.744566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df = sub_df.loc[sub_df.id.str.contains('_image')]\nlen(image_df)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.746627Z","iopub.execute_input":"2021-08-09T15:06:07.752086Z","iopub.status.idle":"2021-08-09T15:06:07.766976Z","shell.execute_reply.started":"2021-08-09T15:06:07.752045Z","shell.execute_reply":"2021-08-09T15:06:07.766002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n\nfrom PIL import Image\n\nfrom shutil import copyfile\nfrom sklearn.model_selection import train_test_split\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.768320Z","iopub.execute_input":"2021-08-09T15:06:07.769049Z","iopub.status.idle":"2021-08-09T15:06:07.779352Z","shell.execute_reply.started":"2021-08-09T15:06:07.769006Z","shell.execute_reply":"2021-08-09T15:06:07.777867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.781164Z","iopub.execute_input":"2021-08-09T15:06:07.781641Z","iopub.status.idle":"2021-08-09T15:06:07.803075Z","shell.execute_reply.started":"2021-08-09T15:06:07.781602Z","shell.execute_reply":"2021-08-09T15:06:07.801483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.804790Z","iopub.execute_input":"2021-08-09T15:06:07.805577Z","iopub.status.idle":"2021-08-09T15:06:07.825645Z","shell.execute_reply.started":"2021-08-09T15:06:07.805539Z","shell.execute_reply":"2021-08-09T15:06:07.824562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Files","metadata":{}},{"cell_type":"code","source":"meta_df=pd.read_csv('/kaggle/input/siim-covid19-resized-to-256px-jpg/meta.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.830390Z","iopub.execute_input":"2021-08-09T15:06:07.831015Z","iopub.status.idle":"2021-08-09T15:06:07.853773Z","shell.execute_reply.started":"2021-08-09T15:06:07.830977Z","shell.execute_reply":"2021-08-09T15:06:07.852800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for study_dir in os.listdir('/kaggle/input/siim-covid19-detection/test'):\n    for series in os.listdir(f'/kaggle/input/siim-covid19-detection/test/{study_dir}'):\n        for image in os.listdir(f'/kaggle/input/siim-covid19-detection/test/{study_dir}/{series}/'):\n            image_id = image[:-4]\n            meta_df.loc[meta_df['image_id'] == image_id, 'study_id'] = study_dir\n        \nmeta_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:07.857349Z","iopub.execute_input":"2021-08-09T15:06:07.859348Z","iopub.status.idle":"2021-08-09T15:06:14.278264Z","shell.execute_reply.started":"2021-08-09T15:06:07.859307Z","shell.execute_reply":"2021-08-09T15:06:14.277412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df_test=meta_df.loc[meta_df['split']=='test']","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:14.279498Z","iopub.execute_input":"2021-08-09T15:06:14.279842Z","iopub.status.idle":"2021-08-09T15:06:14.288969Z","shell.execute_reply.started":"2021-08-09T15:06:14.279801Z","shell.execute_reply":"2021-08-09T15:06:14.288120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del meta_df_test['split']","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:14.290464Z","iopub.execute_input":"2021-08-09T15:06:14.291157Z","iopub.status.idle":"2021-08-09T15:06:14.296906Z","shell.execute_reply.started":"2021-08-09T15:06:14.291119Z","shell.execute_reply":"2021-08-09T15:06:14.295870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create Meta file for Test  dataset","metadata":{}},{"cell_type":"code","source":"meta_df_test","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:14.298504Z","iopub.execute_input":"2021-08-09T15:06:14.299097Z","iopub.status.idle":"2021-08-09T15:06:14.316647Z","shell.execute_reply.started":"2021-08-09T15:06:14.299056Z","shell.execute_reply":"2021-08-09T15:06:14.315863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# YOLO_MODEL_PATH = '../input/yolo-models/yolov5s-e-100-img-512.pt'\nYOLO_MODEL_PATHS = 'kaggle-siim-covid19/exp/weights/best.pt'\n\n\n!python detect.py --weights {weights_dir} \\\n                  --source {TEST_PATH} \\\n                  --img {IMG_SIZE} \\\n                  --conf 0.28 \\\n                  --iou-thres 0.5 \\\n                  --max-det 3 \\\n                  --save-txt \\\n                  --save-conf \\\n                  --exist-ok","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:06:14.317905Z","iopub.execute_input":"2021-08-09T15:06:14.318269Z","iopub.status.idle":"2021-08-09T15:07:07.523087Z","shell.execute_reply.started":"2021-08-09T15:06:14.318233Z","shell.execute_reply":"2021-08-09T15:07:07.522073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('runs/detect/')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:07.526412Z","iopub.execute_input":"2021-08-09T15:07:07.526706Z","iopub.status.idle":"2021-08-09T15:07:07.534410Z","shell.execute_reply.started":"2021-08-09T15:07:07.526673Z","shell.execute_reply":"2021-08-09T15:07:07.533307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport numpy as np\nimport random\nimport cv2\nfrom glob import glob\nfrom tqdm import tqdm\n\nfiles = glob('runs/detect/exp/*')\nfor _ in range(3):\n    row = 4\n    col = 3\n    grid_files = random.sample(files, row*col)\n    images     = []\n    for image_path in tqdm(grid_files):\n        img= cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n        images.append(img)\n\n    fig = plt.figure(figsize=(col*5, row*5))\n    grid = ImageGrid(fig, 111,  # similar to subplot(111)\n                     nrows_ncols=(col, row),  # creates 2x2 grid of axes\n                     axes_pad=0.05,  # pad between axes in inch.\n                     )\n\n    for ax, im in zip(grid, images):\n        # Iterating over the grid returns the Axes.\n        ax.imshow(im)\n        ax.set_xticks([])\n        ax.set_yticks([])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:07.535982Z","iopub.execute_input":"2021-08-09T15:07:07.536453Z","iopub.status.idle":"2021-08-09T15:07:10.902482Z","shell.execute_reply.started":"2021-08-09T15:07:07.536412Z","shell.execute_reply":"2021-08-09T15:07:10.901420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PRED_PATH /kaggle/runs/detect/exp/labels'\nprediction_files = os.listdir(PRED_PATH)\nprint(f'Number of opacity predicted by YOLOv5: {len(prediction_files)}')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:10.903860Z","iopub.execute_input":"2021-08-09T15:07:10.904183Z","iopub.status.idle":"2021-08-09T15:07:10.910560Z","shell.execute_reply.started":"2021-08-09T15:07:10.904149Z","shell.execute_reply":"2021-08-09T15:07:10.909620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport gc\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:10.911904Z","iopub.execute_input":"2021-08-09T15:07:10.912441Z","iopub.status.idle":"2021-08-09T15:07:12.637551Z","shell.execute_reply.started":"2021-08-09T15:07:10.912403Z","shell.execute_reply":"2021-08-09T15:07:12.636616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\n\nCONFIG = dict (\n    seed = 42,\n    num_labels = 4,\n    num_folds = 5,\n    img_width = 256,\n    img_height = 256,\n    batch_size = 8,\n    architecture = \"CNN\",\n    infra = \"GCP\",\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:12.638869Z","iopub.execute_input":"2021-08-09T15:07:12.639249Z","iopub.status.idle":"2021-08-09T15:07:12.645173Z","shell.execute_reply.started":"2021-08-09T15:07:12.639211Z","shell.execute_reply":"2021-08-09T15:07:12.644249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = '/kaggle/input/siim-covid19-resized-to-256px-jpg/test/'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:12.646682Z","iopub.execute_input":"2021-08-09T15:07:12.647072Z","iopub.status.idle":"2021-08-09T15:07:12.656622Z","shell.execute_reply.started":"2021-08-09T15:07:12.647032Z","shell.execute_reply":"2021-08-09T15:07:12.655347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df['path'] = image_df.apply(lambda row: TEST_PATH+row.id.split('_')[0]+'.jpg', axis=1)\nimage_df = image_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:12.658002Z","iopub.execute_input":"2021-08-09T15:07:12.658476Z","iopub.status.idle":"2021-08-09T15:07:12.689351Z","shell.execute_reply.started":"2021-08-09T15:07:12.658425Z","shell.execute_reply":"2021-08-09T15:07:12.688533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:12.690616Z","iopub.execute_input":"2021-08-09T15:07:12.690982Z","iopub.status.idle":"2021-08-09T15:07:12.703672Z","shell.execute_reply.started":"2021-08-09T15:07:12.690946Z","shell.execute_reply":"2021-08-09T15:07:12.702547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df = sub_df.loc[sub_df.id.str.contains('_study')]\nlen(study_df)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:12.705222Z","iopub.execute_input":"2021-08-09T15:07:12.705661Z","iopub.status.idle":"2021-08-09T15:07:12.716711Z","shell.execute_reply.started":"2021-08-09T15:07:12.705591Z","shell.execute_reply":"2021-08-09T15:07:12.715607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks to https://www.kaggle.com/ayuraj/submission-covid19/data","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef decode_image(image):\n    # convert the compressed string to a 3D uint8 tensor\n    image = tf.image.decode_png(image, channels=3)\n    # Normalize image\n    image = tf.image.convert_image_dtype(image, dtype=tf.float32)\n    return image\n\n@tf.function\ndef load_image(df_dict):\n    # Load image\n    image = tf.io.read_file(df_dict['path'])\n    image = decode_image(image)\n    \n    # Resize image\n    image = tf.image.resize(image, (CONFIG['img_height'], CONFIG['img_width']))\n    \n    return image\n\ntestloader = tf.data.Dataset.from_tensor_slices(dict(image_df))\n\ntestloader = (\n    testloader\n    .shuffle(1024)\n    .map(load_image, num_parallel_calls=AUTOTUNE)\n    .batch(CONFIG['batch_size'])\n    .prefetch(AUTOTUNE)\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:12.718468Z","iopub.execute_input":"2021-08-09T15:07:12.718815Z","iopub.status.idle":"2021-08-09T15:07:15.424152Z","shell.execute_reply.started":"2021-08-09T15:07:12.718780Z","shell.execute_reply":"2021-08-09T15:07:15.423245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Model\nSTUDY_MODEL_PATHS = '/kaggle/input/studylevelmodel/SIIM-Study-Level-model/'\nstudy_models = os.listdir(STUDY_MODEL_PATHS)\nstudy_models","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:15.425536Z","iopub.execute_input":"2021-08-09T15:07:15.425877Z","iopub.status.idle":"2021-08-09T15:07:15.446171Z","shell.execute_reply.started":"2021-08-09T15:07:15.425840Z","shell.execute_reply":"2021-08-09T15:07:15.445438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Model for Study _pred","metadata":{}},{"cell_type":"code","source":"! pip install -q efficientnet\nfrom efficientnet.tfkeras import EfficientNetB5","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:15.447507Z","iopub.execute_input":"2021-08-09T15:07:15.447863Z","iopub.status.idle":"2021-08-09T15:07:22.496635Z","shell.execute_reply.started":"2021-08-09T15:07:15.447825Z","shell.execute_reply":"2021-08-09T15:07:22.495589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\nfor model in study_models:\n    # Load model\n    tf.keras.backend.clear_session()\n    model = tf.keras.models.load_model(STUDY_MODEL_PATHS+model)\n    # Prediction\n    tmp = []\n    for img_batch in tqdm(testloader):\n        preds = model.predict(img_batch)\n        tmp.extend(preds)\n        \n    predictions.append(tmp)\n    \n    del model\n    _ = gc.collect()\n    \npredictions = np.mean(predictions, axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:07:22.498344Z","iopub.execute_input":"2021-08-09T15:07:22.498742Z","iopub.status.idle":"2021-08-09T15:09:55.465495Z","shell.execute_reply.started":"2021-08-09T15:07:22.498698Z","shell.execute_reply":"2021-08-09T15:09:55.464545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['0', '1', '2', '3']\nimage_df.loc[:, class_labels] = predictions\nimage_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:55.468409Z","iopub.execute_input":"2021-08-09T15:09:55.468771Z","iopub.status.idle":"2021-08-09T15:09:55.487398Z","shell.execute_reply.started":"2021-08-09T15:09:55.468736Z","shell.execute_reply":"2021-08-09T15:09:55.486525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_to_id = { \n    'negative': 0,\n    'typical': 1,\n    'indeterminate': 2,\n    'atypical': 3}\nid_to_class  = {v:k for k, v in class_to_id.items()}\n\ndef get_study_prediction_string(preds, threshold=0):\n    string = ''\n    for idx in range(4):\n        conf =  preds[idx]\n        if conf>threshold:\n            string+=f'{id_to_class[idx]} {conf:0.2f} 0 0 1 1 '\n    string = string.strip()\n    return string","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:55.488837Z","iopub.execute_input":"2021-08-09T15:09:55.489413Z","iopub.status.idle":"2021-08-09T15:09:55.498374Z","shell.execute_reply.started":"2021-08-09T15:09:55.489371Z","shell.execute_reply":"2021-08-09T15:09:55.497485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:55.505969Z","iopub.execute_input":"2021-08-09T15:09:55.506397Z","iopub.status.idle":"2021-08-09T15:09:55.520397Z","shell.execute_reply.started":"2021-08-09T15:09:55.506369Z","shell.execute_reply":"2021-08-09T15:09:55.519409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:55.522218Z","iopub.execute_input":"2021-08-09T15:09:55.522783Z","iopub.status.idle":"2021-08-09T15:09:55.537511Z","shell.execute_reply.started":"2021-08-09T15:09:55.522743Z","shell.execute_reply":"2021-08-09T15:09:55.536533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_ids = []\npred_strings = []\n\nfor study_id, df in meta_df_test.groupby('study_id'):\n    # accumulate preds for diff images belonging to same study_id\n    tmp_pred = []\n    \n    df = df.reset_index(drop=True)\n    for image_id in df.image_id.values:\n        preds = image_df.loc[image_df.id == image_id+'_image'].values[0]\n        tmp_pred.append(preds[3:])\n    \n    preds = np.mean(tmp_pred, axis=0)\n    pred_string = get_study_prediction_string(preds)\n    pred_strings.append(pred_string)\n    \n    study_ids.append(f'{study_id}_study')\n    \nstudy_df = pd.DataFrame.from_dict({'id': study_ids, 'PredictionString': pred_strings})\nstudy_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:55.539168Z","iopub.execute_input":"2021-08-09T15:09:55.539692Z","iopub.status.idle":"2021-08-09T15:09:56.862676Z","shell.execute_reply.started":"2021-08-09T15:09:55.539607Z","shell.execute_reply":"2021-08-09T15:09:56.861739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The submisison requires xmin, ymin, xmax, ymax format. \n# YOLOv5 returns x_center, y_center, width, height\ndef correct_bbox_format(bboxes):\n    correct_bboxes = []\n    for b in bboxes:\n        xc, yc = int(np.round(b[0]*IMG_SIZE)), int(np.round(b[1]*IMG_SIZE))\n        w, h = int(np.round(b[2]*IMG_SIZE)), int(np.round(b[3]*IMG_SIZE))\n\n        xmin = xc - int(np.round(w/2))\n        ymin = yc - int(np.round(h/2))\n        xmax = xc + int(np.round(w/2))\n        ymax = yc + int(np.round(h/2))\n        \n        correct_bboxes.append([xmin, ymin, xmax, ymax])\n        \n    return correct_bboxes\n\ndef scale_bboxes_to_original(row, bboxes):\n    # Get scaling factor\n    scale_x = IMG_SIZE/row.dim1\n    scale_y = IMG_SIZE/row.dim0\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        xmin, ymin, xmax, ymax = bbox\n        \n        xmin = int(np.round(xmin/scale_x))\n        ymin = int(np.round(ymin/scale_y))\n        xmax = int(np.round(xmax/scale_x))\n        ymax = int(np.round(ymax/scale_y))\n        \n        scaled_bboxes.append([xmin, ymin, xmax, ymax])\n        \n    return scaled_bboxes\n\n# Read the txt file generated by YOLOv5 during inference and extract \n# confidence and bounding box coordinates.\ndef get_conf_bboxes(file_path):\n    confidence = []\n    bboxes = []\n    with open(file_path, 'r') as file:\n        for line in file:\n            preds = line.strip('\\n').split(' ')\n            preds = list(map(float, preds))\n            confidence.append(preds[-1])\n            bboxes.append(preds[1:-1])\n    return confidence, bboxes","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:56.864107Z","iopub.execute_input":"2021-08-09T15:09:56.864472Z","iopub.status.idle":"2021-08-09T15:09:56.876394Z","shell.execute_reply.started":"2021-08-09T15:09:56.864417Z","shell.execute_reply":"2021-08-09T15:09:56.875258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_pred_strings = []\nfor i in tqdm(range(len(image_df))):\n    row = meta_df_test.loc[i]\n    id_name = row.image_id\n    \n    if f'{id_name}.txt' in prediction_files:\n        # opacity label\n        confidence, bboxes = get_conf_bboxes(f'{PRED_PATH}/{id_name}.txt')\n        bboxes = correct_bbox_format(bboxes)\n        ori_bboxes = scale_bboxes_to_original(row, bboxes)\n        \n        pred_string = ''\n        for j, conf in enumerate(confidence):\n            pred_string += f'opacity {conf} ' + ' '.join(map(str, ori_bboxes[j])) + ' '\n        image_pred_strings.append(pred_string[:-1]) \n    else:\n        image_pred_strings.append(\"none 1 0 0 1 1\")","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:56.877673Z","iopub.execute_input":"2021-08-09T15:09:56.878094Z","iopub.status.idle":"2021-08-09T15:09:57.298032Z","shell.execute_reply.started":"2021-08-09T15:09:56.878005Z","shell.execute_reply":"2021-08-09T15:09:57.297052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission File","metadata":{}},{"cell_type":"code","source":"meta_df_test['PredictionString'] = image_pred_strings\nimage_df = meta_df_test[['image_id', 'PredictionString']]\nimage_df.insert(0, 'id', image_df.apply(lambda row: row.image_id+'_image', axis=1))\nimage_df = image_df.drop('image_id', axis=1)\nimage_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:57.299535Z","iopub.execute_input":"2021-08-09T15:09:57.299939Z","iopub.status.idle":"2021-08-09T15:09:57.336649Z","shell.execute_reply.started":"2021-08-09T15:09:57.299899Z","shell.execute_reply":"2021-08-09T15:09:57.335659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf runs","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:57.338029Z","iopub.execute_input":"2021-08-09T15:09:57.338384Z","iopub.status.idle":"2021-08-09T15:09:58.089756Z","shell.execute_reply.started":"2021-08-09T15:09:57.338347Z","shell.execute_reply":"2021-08-09T15:09:58.088675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.concat([study_df, image_df])\nsub_df.to_csv('/kaggle/working/submission.csv', index=False)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2021-08-09T15:09:58.093727Z","iopub.execute_input":"2021-08-09T15:09:58.094036Z","iopub.status.idle":"2021-08-09T15:09:58.126540Z","shell.execute_reply.started":"2021-08-09T15:09:58.094004Z","shell.execute_reply":"2021-08-09T15:09:58.125531Z"},"trusted":true},"execution_count":null,"outputs":[]}]}