{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:06:45.418925Z","iopub.execute_input":"2021-08-04T14:06:45.419261Z","iopub.status.idle":"2021-08-04T14:07:53.410544Z","shell.execute_reply.started":"2021-08-04T14:06:45.419186Z","shell.execute_reply":"2021-08-04T14:07:53.409555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys; \n\npackage_paths = [\n    '../input/omegaconf',\n    '../input/timm-pytorch-image-models/pytorch-image-models-master',\n    '../input/weightedboxesfusion/Weighted-Boxes-Fusion',\n    '../input/efficientdetpytorch/efficientdet-pytorch'\n]\n\nfor pth in package_paths:\n    sys.path.append(pth)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os\nimport time\nimport random\n\nimport numpy as np  # linear algebra!\nimport pandas as pd  # data processing, CSV file I/O (e.g. pd.read_csv)\nimport PIL\n\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import roc_auc_score\n\nimport cv2\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom tqdm import tqdm\nfrom tqdm.contrib.concurrent import process_map\n\nimport torch\nimport torchvision\nfrom torch.utils.data.dataset import Dataset\nimport torch.cuda.amp as amp\n\nimport timm\nfrom ensemble_boxes import *\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nfrom effdet import get_efficientdet_config, EfficientDet, DetBenchPredict\nfrom effdet.efficientdet import HeadNet\nfrom multiprocessing import Pool, cpu_count\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nSIZE = (384, 384)\n\nMODEL_DIR = \"/kaggle/input/classification-training-script1\"\nDATA_DIR = RESIZE_DIR = \"/kaggle/working/\"\nFOLDS = 5\nNUM_CLASSES = 4\nBATCHSIZE = 64\nSEED = 420\nMODEL_NAME = \"tf_efficientnetv2_s\"","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:55.955117Z","iopub.execute_input":"2021-08-04T14:10:55.955487Z","iopub.status.idle":"2021-08-04T14:10:59.962201Z","shell.execute_reply.started":"2021-08-04T14:10:55.955454Z","shell.execute_reply":"2021-08-04T14:10:59.959948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nprint(len(sub_df))\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:59.96341Z","iopub.status.idle":"2021-08-04T14:10:59.963836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df = sub_df.loc[sub_df.id.str.contains('_study')]\nlen(study_df)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:59.965132Z","iopub.status.idle":"2021-08-04T14:10:59.965841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df = sub_df.loc[sub_df.id.str.contains('_image')]\nlen(image_df)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:59.967166Z","iopub.status.idle":"2021-08-04T14:10:59.967988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\ndef resize_xray(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:59.969265Z","iopub.status.idle":"2021-08-04T14:10:59.969947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = f'/kaggle/working/test_{SIZE[0]}x{SIZE[1]}'\n\nos.makedirs(TEST_PATH, exist_ok=True)\nfilenames = glob.glob(f'/kaggle/input/siim-covid19-detection/test/*/*/*.dcm')\n\ndef persist_image(path):\n    xray = read_xray(path)\n    im = resize_xray(xray, size=SIZE[0])\n    fname = os.path.basename(os.path.splitext(path)[-2])\n    jpg_fname = os.path.join(TEST_PATH, \"{}.jpg\".format(fname))\n    im.save(jpg_fname)\n    return [fname,xray.shape[0],xray.shape[1]]\nwith Pool(cpu_count()) as pool:\n    img_metadata = pool.map(persist_image,filenames)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:59.971198Z","iopub.status.idle":"2021-08-04T14:10:59.971869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of test images: {len(os.listdir(TEST_PATH))}')","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:10:59.973243Z","iopub.status.idle":"2021-08-04T14:10:59.973969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping = pd.DataFrame(img_metadata,columns = ['image_id','dim0','dim1'])\n\n# Associate image-level id with study-level ids.\n# Note that a study-level might have more than one image-level ids.\nfor study_dir in os.listdir('../input/siim-covid19-detection/test'):\n    for series in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}'):\n        for image in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}/{series}/'):\n            image_id = image[:-4]\n            test_imgs_study_mapping.loc[test_imgs_study_mapping['image_id'] == image_id, 'study_id'] = study_dir\n        \ntest_imgs_study_mapping.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-04T14:11:00.942674Z","iopub.execute_input":"2021-08-04T14:11:00.943047Z","iopub.status.idle":"2021-08-04T14:11:00.964797Z","shell.execute_reply.started":"2021-08-04T14:11:00.943013Z","shell.execute_reply":"2021-08-04T14:11:00.963592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !mkdir \"/kaggle/working/test_{SIZE[0]}x{SIZE[1]}\"\n# !tar -xzf \"/kaggle/input/train-{SIZE[0]}x{SIZE[1]}/test_{SIZE[0]}x{SIZE[1]}.tar.gz\" -C \"/kaggle/working/test_{SIZE[0]}x{SIZE[1]}\" .","metadata":{"execution":{"iopub.status.busy":"2021-08-04T09:20:23.172162Z","iopub.execute_input":"2021-08-04T09:20:23.172503Z","iopub.status.idle":"2021-08-04T09:20:23.176212Z","shell.execute_reply.started":"2021-08-04T09:20:23.17247Z","shell.execute_reply":"2021-08-04T09:20:23.175317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make results reproducible\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.benchmark = False\n    torch.backends.cudnn.deterministic = True\n\n\nseed_everything(SEED)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T09:20:23.178313Z","iopub.execute_input":"2021-08-04T09:20:23.178866Z","iopub.status.idle":"2021-08-04T09:20:23.195022Z","shell.execute_reply.started":"2021-08-04T09:20:23.17875Z","shell.execute_reply":"2021-08-04T09:20:23.193868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_pattern = f\"{MODEL_NAME}\"\nmodelnames = glob.glob(os.path.join(MODEL_DIR, f\"{MODEL_NAME}*\", \"*.pth\"))\nprint(f\"{len(modelnames)}: Models Found\")","metadata":{"execution":{"iopub.status.busy":"2021-08-04T09:20:23.196561Z","iopub.execute_input":"2021-08-04T09:20:23.19694Z","iopub.status.idle":"2021-08-04T09:20:23.254382Z","shell.execute_reply.started":"2021-08-04T09:20:23.19691Z","shell.execute_reply":"2021-08-04T09:20:23.253309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_imgs_study_mapping = pd.read_csv(\"/kaggle/input/test-image-to-study-mapping/test_study_id_to_image_mapping.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T08:25:21.04297Z","iopub.execute_input":"2021-08-03T08:25:21.043304Z","iopub.status.idle":"2021-08-03T08:25:21.048048Z","shell.execute_reply.started":"2021-08-03T08:25:21.043275Z","shell.execute_reply":"2021-08-03T08:25:21.046615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TEST_DATA_PATH = os.path.join(RESIZE_DIR, f\"test_{SIZE[0]}x{SIZE[1]}\")\nprint(\"Test Data Path {}\".format(TEST_PATH))\ndef get_img_path(row):\n    study_id = row[\"study_id\"]\n    img_id = row[\"image_id\"]\n    paths = glob.glob(os.path.join(TEST_PATH, \"{}*.jpg\".format(img_id)))\n    for path in paths:\n        if img_id in path:\n            return path\n    return None","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:05:42.054891Z","iopub.execute_input":"2021-08-04T12:05:42.055286Z","iopub.status.idle":"2021-08-04T12:05:42.062408Z","shell.execute_reply.started":"2021-08-04T12:05:42.055249Z","shell.execute_reply":"2021-08-04T12:05:42.06081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping[\"path\"] = test_imgs_study_mapping.apply(get_img_path, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:05:42.06382Z","iopub.execute_input":"2021-08-04T12:05:42.064231Z","iopub.status.idle":"2021-08-04T12:05:44.872291Z","shell.execute_reply.started":"2021-08-04T12:05:42.064191Z","shell.execute_reply":"2021-08-04T12:05:44.87126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping[test_imgs_study_mapping[\"path\"].isna()]","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:05:44.873814Z","iopub.execute_input":"2021-08-04T12:05:44.874278Z","iopub.status.idle":"2021-08-04T12:05:44.892961Z","shell.execute_reply.started":"2021-08-04T12:05:44.874231Z","shell.execute_reply":"2021-08-04T12:05:44.891588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:05:44.894518Z","iopub.execute_input":"2021-08-04T12:05:44.895099Z","iopub.status.idle":"2021-08-04T12:05:44.914808Z","shell.execute_reply.started":"2021-08-04T12:05:44.895054Z","shell.execute_reply":"2021-08-04T12:05:44.913958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_model(modelname):\n    summary = torch.load(modelname)\n    print(f\"Loaded model {modelname}\")\n    print(f\"Epoch {summary['epoch']}\")\n    print(f\"Map@2 {summary['map_at_2']}\")\n    print(f\"AUC@2 {summary['auc']}\")\n    model = timm.create_model(model_name=MODEL_NAME, pretrained=False, in_chans=3)\n    model.classifier = torch.nn.Linear(\n        in_features=model.classifier.in_features, out_features=NUM_CLASSES\n    )\n    model.load_state_dict(summary[\"state_dict\"], strict=True)\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-08-03T09:28:32.740635Z","iopub.execute_input":"2021-08-03T09:28:32.741048Z","iopub.status.idle":"2021-08-03T09:28:32.749903Z","shell.execute_reply.started":"2021-08-03T09:28:32.741018Z","shell.execute_reply":"2021-08-03T09:28:32.748616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class XRayDatasetFromDF(Dataset):\n    def __init__(self, df, train=True, augment=True, normalize=False, size=(384, 384)):\n        self.df = df\n        self.name_to_label_map = {\n            \"Negative\": 0,\n            \"Typical\": 1,\n            \"Indeterminate\": 2,\n            \"Atypical\": 3,\n        }\n        self.study_ids = df.index.sort_values()\n        self.path_suffix = (\n            os.path.join(DATA_DIR, \"train\") if train else os.path.join(DATA_DIR, \"test\")\n        )\n        self._train = train\n        self._augment = augment\n        self._normalize = normalize\n        self._size = size\n        self._transform_list = [\n            # A.Resize(size[0], size[1], p=1)\n        ]\n\n        if self._augment:\n            self._transform_list.extend(\n                [\n                    A.VerticalFlip(p=0.5),\n                    A.HorizontalFlip(p=0.5),\n                    A.ShiftScaleRotate(\n                        scale_limit=0.20,\n                        rotate_limit=10,\n                        shift_limit=0.1,\n                        p=0.5,\n                        border_mode=cv2.BORDER_CONSTANT,\n                        value=0,\n                    ),\n                    A.RandomBrightnessContrast(p=0.5),\n                    A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                    ToTensorV2(),\n                ]\n            )\n        elif self._normalize and not self._augment:  # test mode\n            self._transform_list.extend(\n                [\n                    A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                    ToTensorV2(),\n                ]\n            )\n        self._transforms = A.Compose(self._transform_list)\n\n    def __len__(self):\n        return len(self.study_ids)\n\n    def assign_label(self, row):\n        for k in self.name_to_label_map:\n            if row[k]:\n                return self.name_to_label_map[k]\n\n    def __getitem__(self, idx):\n        study_id = self.study_ids[idx]\n        study_imgs = self.df.loc[study_id]\n        path = None\n        label = None\n\n        path = study_imgs[\"path\"]\n        label = study_imgs[\"int_label\"] if self._train else -1\n\n        # ideally, we'd clean up the df,\n        # but may be we use it to produce predictions as well.\n        dicom_arr = (\n            cv2.imread(path)\n            if path.endswith(\".jpg\")\n            else dicom2array(path, size=self._size)\n        )\n        img = cv2.cvtColor(dicom_arr, cv2.COLOR_BGR2RGB)\n        img = self._transforms(image=img)[\"image\"]\n\n        return img, label","metadata":{"execution":{"iopub.status.busy":"2021-08-03T09:28:34.179947Z","iopub.execute_input":"2021-08-03T09:28:34.180368Z","iopub.status.idle":"2021-08-03T09:28:34.19675Z","shell.execute_reply.started":"2021-08-03T09:28:34.180336Z","shell.execute_reply":"2021-08-03T09:28:34.195309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:05:44.918022Z","iopub.execute_input":"2021-08-04T12:05:44.91835Z","iopub.status.idle":"2021-08-04T12:05:44.933122Z","shell.execute_reply.started":"2021-08-04T12:05:44.918321Z","shell.execute_reply":"2021-08-04T12:05:44.932224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = XRayDatasetFromDF(df=test_imgs_study_mapping, train=False, augment=True, normalize=False, size=SIZE)\ntest_dl = torch.utils.data.DataLoader(\n        dataset=test_ds,\n        batch_size=BATCHSIZE * 2,\n        pin_memory=True,\n        num_workers=8,\n        drop_last=False,\n        shuffle=False,\n        prefetch_factor=8,\n    )","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:05:44.934854Z","iopub.execute_input":"2021-08-04T12:05:44.935415Z","iopub.status.idle":"2021-08-04T12:05:45.124741Z","shell.execute_reply.started":"2021-08-04T12:05:44.935376Z","shell.execute_reply":"2021-08-04T12:05:45.122977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(model, slide_dl, tta_times=10):\n    sample_size = len(slide_dl.dataset)\n    print(\"Predicting on {} Images {} times\".format(sample_size, tta_times))\n    probs = np.zeros((sample_size, NUM_CLASSES))\n    loss_sum = 0\n\n    loss_fn = torch.nn.BCEWithLogitsLoss(reduction=\"none\").to(dev)\n\n    for i in range(tta_times):\n\n        offset = 0\n\n        for i, grid in enumerate(tqdm(slide_dl)):\n            with torch.no_grad():\n                img, _ = grid\n                curr_batch_size = img.shape[0]\n\n                pred = model(img.to(dev))\n                # remove the redundant dimension added by\n                # pytorch's collate_fn\n\n                prob = pred.softmax(dim=1)\n                probs[offset : offset + curr_batch_size, :] += prob.cpu().numpy()\n                offset += curr_batch_size\n\n\n    return probs / tta_times","metadata":{"execution":{"iopub.status.busy":"2021-07-22T07:42:03.725959Z","iopub.execute_input":"2021-07-22T07:42:03.726414Z","iopub.status.idle":"2021-07-22T07:42:03.737429Z","shell.execute_reply.started":"2021-07-22T07:42:03.726383Z","shell.execute_reply":"2021-07-22T07:42:03.736209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dev = torch.device(\"cuda\") if torch.cuda.is_available() else torch.device(\"cpu\")\nprint(dev)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:42.03451Z","iopub.execute_input":"2021-08-04T12:06:42.034888Z","iopub.status.idle":"2021-08-04T12:06:42.040909Z","shell.execute_reply.started":"2021-08-04T12:06:42.034853Z","shell.execute_reply":"2021-08-04T12:06:42.039496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\nfor modelname in modelnames:\n    model = load_model(modelname)\n    model = model.to(dev)\n    model = model.eval()\n    probs = predict(model, test_dl)\n    # probs = torch.from_numpy(probs)\n    # probs = probs.softmax(dim=1)\n\n    for k in test_ds.name_to_label_map:\n        col_idx = test_ds.name_to_label_map[k]\n        test_imgs_study_mapping[k] = probs[:, col_idx]\n    test_imgs_study_mapping[\"modelname\"] = modelname\n    predictions.append(test_imgs_study_mapping.copy())","metadata":{"execution":{"iopub.status.busy":"2021-07-22T07:42:03.818961Z","iopub.execute_input":"2021-07-22T07:42:03.819665Z","iopub.status.idle":"2021-07-22T08:11:15.257104Z","shell.execute_reply.started":"2021-07-22T07:42:03.819617Z","shell.execute_reply":"2021-07-22T08:11:15.255125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df = pd.concat(predictions)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.260384Z","iopub.execute_input":"2021-07-22T08:11:15.260833Z","iopub.status.idle":"2021-07-22T08:11:15.285717Z","shell.execute_reply.started":"2021-07-22T08:11:15.260781Z","shell.execute_reply":"2021-07-22T08:11:15.284541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_predictions_df = predictions_df.groupby(\"study_id\").agg({\n    \"Negative\":\"mean\",\n    \"Typical\":\"mean\",\n    \"Indeterminate\":\"mean\",\n    \"Atypical\":\"mean\"\n}).reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.287439Z","iopub.execute_input":"2021-07-22T08:11:15.287924Z","iopub.status.idle":"2021-07-22T08:11:15.319591Z","shell.execute_reply.started":"2021-07-22T08:11:15.28787Z","shell.execute_reply":"2021-07-22T08:11:15.318405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_predictions_df[\"id\"] = mean_predictions_df[\"study_id\"] + \"_study\"","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.321407Z","iopub.execute_input":"2021-07-22T08:11:15.321877Z","iopub.status.idle":"2021-07-22T08:11:15.383615Z","shell.execute_reply.started":"2021-07-22T08:11:15.321819Z","shell.execute_reply":"2021-07-22T08:11:15.382475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_predictions_df","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.385335Z","iopub.execute_input":"2021-07-22T08:11:15.386108Z","iopub.status.idle":"2021-07-22T08:11:15.414199Z","shell.execute_reply.started":"2021-07-22T08:11:15.386061Z","shell.execute_reply":"2021-07-22T08:11:15.413132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"OPBB = \"0 0 1 1\"\ndef generate_cls_prediction_strings(row):\n    predictions = []\n\n    for k in test_ds.name_to_label_map:\n        \n        p_k = row[k]\n        predictions.append(k.lower())\n        predictions.append(str(p_k))\n        predictions.append(OPBB)\n    return \" \".join(predictions)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.416009Z","iopub.execute_input":"2021-07-22T08:11:15.416767Z","iopub.status.idle":"2021-07-22T08:11:15.42425Z","shell.execute_reply.started":"2021-07-22T08:11:15.41672Z","shell.execute_reply":"2021-07-22T08:11:15.423146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_predictions_df[\"PredictionString\"] = mean_predictions_df.apply(generate_cls_prediction_strings, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.426148Z","iopub.execute_input":"2021-07-22T08:11:15.426661Z","iopub.status.idle":"2021-07-22T08:11:15.491914Z","shell.execute_reply.started":"2021-07-22T08:11:15.426603Z","shell.execute_reply":"2021-07-22T08:11:15.490817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cls_submission_df = mean_predictions_df[[\"id\", \"PredictionString\"]]","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:15.493524Z","iopub.execute_input":"2021-07-22T08:11:15.494003Z","iopub.status.idle":"2021-07-22T08:11:15.502648Z","shell.execute_reply.started":"2021-07-22T08:11:15.493938Z","shell.execute_reply":"2021-07-22T08:11:15.501223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_DIR = '/kaggle/input/covid19detectionefficientdet'\nMODEL_NAME = 'tf_efficientdet_d5'\nSIZE = (512,512)\nBATCHSIZE = 16\nmodel_pattern = f\"{MODEL_NAME}\"\nmodelnames = glob.glob(os.path.join(MODEL_DIR, f\"{MODEL_NAME}*\", \"*.pth\"))\nprint(f\"{len(modelnames)}: Models Found\")","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:03.855715Z","iopub.execute_input":"2021-08-04T12:06:03.856171Z","iopub.status.idle":"2021-08-04T12:06:03.897089Z","shell.execute_reply.started":"2021-08-04T12:06:03.856131Z","shell.execute_reply":"2021-08-04T12:06:03.896067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = f'/kaggle/working/test_{SIZE[0]}x{SIZE[1]}'\n\nos.makedirs(TEST_PATH, exist_ok=True)\nfilenames = glob.glob(f'/kaggle/input/siim-covid19-detection/test/*/*/*.dcm')\n\ndef persist_image(path):\n    xray = read_xray(path)\n    im = resize_xray(xray, size=SIZE[0])\n    fname = os.path.basename(os.path.splitext(path)[-2])\n    jpg_fname = os.path.join(TEST_PATH, \"{}.jpg\".format(fname))\n    im.save(jpg_fname)\n    return [fname,xray.shape[0],xray.shape[1]]\nwith Pool(cpu_count()) as pool:\n    img_metadata = pool.map(persist_image,filenames)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping = pd.DataFrame(img_metadata,columns = ['image_id','dim0','dim1'])\n\n# Associate image-level id with study-level ids.\n# Note that a study-level might have more than one image-level ids.\nfor study_dir in os.listdir('../input/siim-covid19-detection/test'):\n    for series in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}'):\n        for image in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}/{series}/'):\n            image_id = image[:-4]\n            test_imgs_study_mapping.loc[test_imgs_study_mapping['image_id'] == image_id, 'study_id'] = study_dir\n        \ntest_imgs_study_mapping.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Test Data Path {}\".format(TEST_PATH))\ndef get_img_path(row):\n    study_id = row[\"study_id\"]\n    img_id = row[\"image_id\"]\n    paths = glob.glob(os.path.join(TEST_PATH, \"{}*.jpg\".format(img_id)))\n    for path in paths:\n        if img_id in path:\n            return path\n    return None","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_imgs_study_mapping[\"path\"] = test_imgs_study_mapping.apply(get_img_path, axis=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class XRayDatasetFromDFObject(Dataset):\n    def __init__(\n        self,\n        df,\n        train=False,\n        predict=True,\n        augment=True,\n        data_dir=os.path.join(DATA_DIR, \"test\"),\n        size=SIZE,\n    ):\n        self.df = df\n        self.label_list = [\"none\", \"opacity\"]\n        self.ids = df.index.sort_values()#[:100]\n        self.path_suffix = data_dir\n        self._augment = augment\n        self._train = train\n        self._predict = predict\n        self._size = size\n        self._transform_list = [\n            # A.Resize(size[0], size[1], p=1)\n        ]\n\n        if self._augment:\n            self._transform_list.extend(\n                [\n                    A.VerticalFlip(p=0.5),\n                    A.HorizontalFlip(p=0.5),\n                    A.ShiftScaleRotate(\n                        scale_limit=0.20,\n                        rotate_limit=10,\n                        shift_limit=0.1,\n                        p=0.5,\n                        border_mode=cv2.BORDER_CONSTANT,\n                        value=0,\n                    ),\n                    A.RandomBrightnessContrast(p=0.5),\n                    # A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                    # ToTensorV2(),\n                ]\n            )\n\n        if self._train or self._predict:\n            self._transform_list.extend(\n                [\n                    A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n                    ToTensorV2(),\n                ]\n            )\n\n        if self._transform_list:\n\n            self._transforms = A.Compose(\n                self._transform_list,\n            )\n\n    def __len__(self):\n        return len(self.ids)\n\n    def draw_bbox_idx(self, idx):\n        img_id = self.ids[idx]\n        row = self.df.loc[img_id]\n        print(img_id)\n        image = PIL.Image.open(row[\"path\"])\n        scaled_w = image.width\n        scaled_h = image.height\n        print((scaled_w, scaled_h))\n        if pd.notna(row[\"boxes\"]):\n            boxes = eval(row[\"boxes\"])\n            draw = PIL.ImageDraw.Draw(image)\n            for box in boxes:\n                box[\"x\"] = box[\"x\"] / row[\"width\"]\n                box[\"y\"] = box[\"y\"] / row[\"height\"]\n                box[\"width\"] = box[\"width\"] / row[\"width\"]\n                box[\"height\"] = box[\"height\"] / row[\"height\"]\n                draw.rectangle(\n                    [\n                        box[\"x\"] * scaled_w,\n                        box[\"y\"] * scaled_h,\n                        (box[\"x\"] + box[\"width\"]) * scaled_w,\n                        (box[\"y\"] + box[\"height\"]) * scaled_h,\n                    ]\n                )\n        return image\n\n    def _yolo_to_voc_format(self, yolo_bboxes):\n        # takes in bounding boxes of the yolo format\n        # converts them to voc format.\n        # yolo format (x_c, y_c, width, height) normalized to 0, 1 by dividing by image dims.\n        # voc format (x_min, y_min, x_max, y_max), unnormalized.\n        scaled_w = self._size[1]\n        scaled_h = self._size[0]\n        bboxes_voc = torch.zeros_like(yolo_bboxes)\n        # x_min = (x_c - width / 2) * scaled_w\n        bboxes_voc[:, 0] = (yolo_bboxes[:, 0] - yolo_bboxes[:, 2] / 2) * scaled_w\n        bboxes_voc[:, 1] = (yolo_bboxes[:, 1] - yolo_bboxes[:, 3] / 2) * scaled_h\n        bboxes_voc[:, 2] = bboxes_voc[:, 0] + yolo_bboxes[:, 2] * scaled_w\n        bboxes_voc[:, 3] = bboxes_voc[:, 1] + yolo_bboxes[:, 3] * scaled_h\n\n        return bboxes_voc\n\n    def draw_bbox_img(self, image, bboxes):\n        image = PIL.Image.fromarray(image)\n        draw = PIL.ImageDraw.Draw(image)\n        for bbox in bboxes:\n            # x_c = bbox[0]\n            # y_c = bbox[1]\n            # width = bbox[2]\n            # height = bbox[3]\n            # x_1 = (x_c - width / 2) * image.width\n            # y_1 = (y_c - height / 2) * image.height\n            # x_2 = x_1 + width * image.width\n            # y_2 = y_1 + height * image.height\n            # draw.rectangle([x_1, y_1, x_2, y_2])\n            draw.rectangle([bbox[0], bbox[1], bbox[2], bbox[3]])\n#         print(f\"Number of boxes{len(label)}\")\n        return image\n\n    def __getitem__(self, idx):\n        img_id = self.ids[idx]\n        row = self.df.loc[img_id]\n\n        path = row[\"path\"]\n        image_id = row[\"image_id\"]\n        # ideally, we'd clean up the df,\n        # but may be we use it to produce predictions as well.\n        dicom_arr = (\n            cv2.imread(path)\n            if path.endswith(\".jpg\")\n            else dicom2array(path, size=self._size)\n        )\n        img = cv2.cvtColor(dicom_arr, cv2.COLOR_BGR2RGB)\n        image_and_labels = {}\n        if self._augment or (self._train or self._predict):\n            image_and_labels = self._transforms(image=img)\n        image_and_labels[\"imageid\"] = image_id\n        return image_and_labels","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:05.561086Z","iopub.execute_input":"2021-08-04T12:06:05.561469Z","iopub.status.idle":"2021-08-04T12:06:05.586051Z","shell.execute_reply.started":"2021-08-04T12:06:05.561438Z","shell.execute_reply":"2021-08-04T12:06:05.584698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = XRayDatasetFromDFObject(test_imgs_study_mapping)\ntest_dl = torch.utils.data.DataLoader(\n        dataset=test_ds,\n        batch_size=BATCHSIZE * 2,\n        pin_memory=True,\n        num_workers=8,\n        drop_last=False,\n        shuffle=False,\n        prefetch_factor=8,\n    )","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:07.375588Z","iopub.execute_input":"2021-08-04T12:06:07.375931Z","iopub.status.idle":"2021-08-04T12:06:07.383107Z","shell.execute_reply.started":"2021-08-04T12:06:07.375898Z","shell.execute_reply":"2021-08-04T12:06:07.382134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_net(modelname):\n    summary = torch.load(modelname)\n    config = get_efficientdet_config(MODEL_NAME)\n\n    config.image_size = SIZE\n    config.norm_kwargs = dict(eps=0.001, momentum=0.01)\n\n    net = EfficientDet(config, pretrained_backbone=True)\n    net.reset_head(num_classes=2)\n    net.class_net = HeadNet(config, num_outputs=config.num_classes)\n    net.load_state_dict(summary[\"state_dict\"], strict=True)\n\n    return DetBenchPredict(net)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:09.380749Z","iopub.execute_input":"2021-08-04T12:06:09.38112Z","iopub.status.idle":"2021-08-04T12:06:09.387334Z","shell.execute_reply.started":"2021-08-04T12:06:09.381086Z","shell.execute_reply":"2021-08-04T12:06:09.38615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predictObj(dataloader,header):\n    num_val = len(dataloader)\n    dataiter = iter(dataloader)\n    detections = []\n    with torch.no_grad():\n        for i in tqdm(range(num_val)):\n            data = next(dataiter)\n            images = data[\"image\"]\n            batchsize = images.shape[0]\n            images = images.to(dev)\n\n            batched_detections = net(images)\n  \n            offset = 0\n            ids = data[\"imageid\"]\n            label_list = dataloader.dataset.label_list\n            all_preds = []\n            \n            num_imgs, num_dets_per_img = (\n                batched_detections.shape[0],\n                batched_detections.shape[1],\n            )\n            img_ids = []\n\n            for i in range(num_imgs):\n                img_ids.extend([ids[offset]] * num_dets_per_img)\n                offset += 1\n\n            batched_preds_df = pd.DataFrame.from_records(\n                batched_detections.view(num_imgs * num_dets_per_img, -1).tolist(),\n                columns=[\"XMin\", \"YMin\", \"XMax\", \"YMax\", \"Conf\", \"LabelName\"],\n            )\n            batched_preds_df[\"ImageID\"] = img_ids\n            batched_preds_df[\"LabelName\"] = batched_preds_df.apply(\n                lambda x: label_list[int(x[\"LabelName\"] - 1)], axis=1\n            )\n            all_preds.append(batched_preds_df)\n            all_preds_df = pd.concat(all_preds)\n            with open('ImageLevelPredictions.csv','a') as f:\n                all_preds_df.to_csv(f,header=header, index=False)\n            header = False","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:10.405017Z","iopub.execute_input":"2021-08-04T12:06:10.405379Z","iopub.status.idle":"2021-08-04T12:06:10.41598Z","shell.execute_reply.started":"2021-08-04T12:06:10.405348Z","shell.execute_reply":"2021-08-04T12:06:10.414445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"header = True\nfor modelname in tqdm(modelnames):\n    net = get_net(modelname)\n    net = net.to(dev)\n    net = net.eval()\n    predictObj(test_dl,header)\n    header = False","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:06:49.180426Z","iopub.execute_input":"2021-08-04T12:06:49.18078Z","iopub.status.idle":"2021-08-04T12:15:39.191775Z","shell.execute_reply.started":"2021-08-04T12:06:49.180748Z","shell.execute_reply":"2021-08-04T12:15:39.190538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"modelPreds = pd.read_csv('/kaggle/working/ImageLevelPredictions.csv')\nprint(len(modelPreds))\nmodelPreds.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:19:14.13131Z","iopub.execute_input":"2021-08-04T12:19:14.131743Z","iopub.status.idle":"2021-08-04T12:19:15.564556Z","shell.execute_reply.started":"2021-08-04T12:19:14.1317Z","shell.execute_reply":"2021-08-04T12:19:15.563411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(modelPreds.ImageID.unique())","metadata":{"execution":{"iopub.status.busy":"2021-08-04T12:19:20.72496Z","iopub.execute_input":"2021-08-04T12:19:20.725332Z","iopub.status.idle":"2021-08-04T12:19:20.816503Z","shell.execute_reply.started":"2021-08-04T12:19:20.725297Z","shell.execute_reply":"2021-08-04T12:19:20.815315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# notweird = modelPreds[(modelPreds[\"XMin\"] > 0) & (modelPreds[\"YMin\"] > 0) & (modelPreds[\"XMax\"] > 0) & (modelPreds[\"YMax\"] > 0)].groupby([\"ImageID\", \"LabelName\"]).agg(\n# {\n#     \"Unnamed: 0\":\"count\",\n#     \"Conf\":\"mean\",\n# }\n# )","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:54:39.991276Z","iopub.execute_input":"2021-08-03T13:54:39.991689Z","iopub.status.idle":"2021-08-03T13:54:40.17418Z","shell.execute_reply.started":"2021-08-03T13:54:39.991655Z","shell.execute_reply":"2021-08-03T13:54:40.173015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# notweird.reset_index(inplace=True)\n# sns.kdeplot(data=notweird, x=\"Conf\", hue=\"LabelName\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:54:41.148907Z","iopub.execute_input":"2021-08-03T13:54:41.149339Z","iopub.status.idle":"2021-08-03T13:54:41.439794Z","shell.execute_reply.started":"2021-08-03T13:54:41.149302Z","shell.execute_reply":"2021-08-03T13:54:41.438645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modelPreds.sort_values([\"Conf\", \"ImageID\"], ascending=False, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:54:42.566561Z","iopub.execute_input":"2021-08-03T13:54:42.566928Z","iopub.status.idle":"2021-08-03T13:54:43.391476Z","shell.execute_reply.started":"2021-08-03T13:54:42.566895Z","shell.execute_reply":"2021-08-03T13:54:43.390287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modelPreds[modelPreds['ImageID'] == 'ffc66893d9ea'].head(20)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:54:45.390348Z","iopub.execute_input":"2021-08-03T13:54:45.390775Z","iopub.status.idle":"2021-08-03T13:54:45.526486Z","shell.execute_reply.started":"2021-08-03T13:54:45.39073Z","shell.execute_reply":"2021-08-03T13:54:45.525336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(modelPreds[modelPreds['LabelName'] == 'none'][modelPreds[modelPreds['LabelName'] == 'none'][\"Conf\"] > 0.75].ImageID.unique())","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:59:28.65617Z","iopub.execute_input":"2021-08-03T13:59:28.656585Z","iopub.status.idle":"2021-08-03T13:59:28.967498Z","shell.execute_reply.started":"2021-08-03T13:59:28.656553Z","shell.execute_reply":"2021-08-03T13:59:28.966125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1992","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# modelPreds[modelPreds['LabelName'] == 'opacity']","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:55:25.836162Z","iopub.execute_input":"2021-08-03T13:55:25.836608Z","iopub.status.idle":"2021-08-03T13:55:26.023786Z","shell.execute_reply.started":"2021-08-03T13:55:25.836573Z","shell.execute_reply":"2021-08-03T13:55:26.022362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weird = modelPreds[(modelPreds[\"XMin\"] < 0) | (modelPreds[\"YMin\"] < 0) | (modelPreds[\"XMax\"] < 0) | (modelPreds[\"YMax\"] < 0)].groupby([\"ImageID\", \"LabelName\"]).agg(\n# {\n#     \"Unnamed: 0\":\"count\",\n#     \"Conf\":\"mean\",\n# }\n# )\n","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:44:52.914603Z","iopub.execute_input":"2021-08-03T13:44:52.915041Z","iopub.status.idle":"2021-08-03T13:44:52.989203Z","shell.execute_reply.started":"2021-08-03T13:44:52.91501Z","shell.execute_reply":"2021-08-03T13:44:52.987958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weird.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:44:54.488191Z","iopub.execute_input":"2021-08-03T13:44:54.488767Z","iopub.status.idle":"2021-08-03T13:44:54.502148Z","shell.execute_reply.started":"2021-08-03T13:44:54.48872Z","shell.execute_reply":"2021-08-03T13:44:54.500118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:44:56.691504Z","iopub.execute_input":"2021-08-03T13:44:56.691911Z","iopub.status.idle":"2021-08-03T13:44:56.698073Z","shell.execute_reply.started":"2021-08-03T13:44:56.691862Z","shell.execute_reply":"2021-08-03T13:44:56.696823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.kdeplot(data=weird, x=\"Conf\", hue=\"LabelName\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:44:57.119347Z","iopub.execute_input":"2021-08-03T13:44:57.119738Z","iopub.status.idle":"2021-08-03T13:44:57.408357Z","shell.execute_reply.started":"2021-08-03T13:44:57.119704Z","shell.execute_reply":"2021-08-03T13:44:57.406765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(modelPreds.ImageID.unique())","metadata":{"execution":{"iopub.status.busy":"2021-08-03T12:45:06.694038Z","iopub.execute_input":"2021-08-03T12:45:06.69444Z","iopub.status.idle":"2021-08-03T12:45:06.765862Z","shell.execute_reply.started":"2021-08-03T12:45:06.694405Z","shell.execute_reply":"2021-08-03T12:45:06.764401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_bboxes_to_original(row, bboxes):\n    # Get scaling factor\n    scale_x = 512\n    scale_y = 512\n    \n    scaled_bboxes = []\n    for bbox in bboxes:\n        xmin, ymin, xmax, ymax = bbox\n        \n        xmin = int(np.round(xmin*scale_x))\n        ymin = int(np.round(ymin*scale_y))\n        xmax = int(np.round(xmax*scale_x))\n        ymax = int(np.round(ymax*scale_y))\n        \n        scaled_bboxes.append([xmin, ymin, xmax, ymax])\n        \n    return scaled_bboxes\ndef clipCoord(coord):\n    if coord < 0:\n        coord = 0.01\n    elif coord > 1:\n        coord = 0.99\n    return coord\n\niou_thr = 0.5\nskip_box_thr = 0.0001\nsigma = 0.1\nimage_id = modelPreds.ImageID.unique()\nPredString = []\npredarr = []\nfor img in tqdm(image_id):\n    predStr = ''\n    oneImagedf = modelPreds[modelPreds.ImageID == img]\n    boxes = []\n    scores = []\n    labels = []\n    for index, row in oneImagedf.iterrows():\n        box = [clipCoord(row.XMin/SIZE[0]),clipCoord(row.YMin/SIZE[0]),clipCoord(row.XMax/SIZE[0]),clipCoord(row.YMax/SIZE[0])]\n        score = row.Conf\n        label = 0 if row.LabelName=='none' else 1\n        boxes.append(box)\n        scores.append(score)\n        labels.append(label)\n    boxes_wbf, scores_wbf, labels_wbf = weighted_boxes_fusion([boxes], [scores], [labels], weights=None, iou_thr=iou_thr, skip_box_thr=skip_box_thr)\n    row = test_imgs_study_mapping.loc[test_imgs_study_mapping.image_id == img]\n    orig_bboxs = scale_bboxes_to_original(row,boxes_wbf)\n    for i,orig_bb in enumerate(orig_bboxs):\n        LabelName = 'opacity' if labels_wbf[i] == 1.0 else 'none'\n#         predStr += f'{LabelName} {scores_wbf[i]} ' + ' '.join(map(str, orig_bb)) + ' '\n        predarr.append([img,LabelName,scores_wbf[i],orig_bb[0],orig_bb[1],orig_bb[2],orig_bb[3]])\n#     PredString.append(predStr)\n","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:25:53.084829Z","iopub.execute_input":"2021-08-04T13:25:53.085271Z","iopub.status.idle":"2021-08-04T13:43:08.055573Z","shell.execute_reply.started":"2021-08-04T13:25:53.085233Z","shell.execute_reply":"2021-08-04T13:43:08.054678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"modelPreds[modelPreds[\"XMin\"].isna() | modelPreds[\"YMin\"].isna() | modelPreds[\"XMax\"].isna() | modelPreds[\"YMax\"].isna()]","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:43:31.942089Z","iopub.execute_input":"2021-08-04T13:43:31.942431Z","iopub.status.idle":"2021-08-04T13:43:31.965173Z","shell.execute_reply.started":"2021-08-04T13:43:31.942401Z","shell.execute_reply":"2021-08-04T13:43:31.964192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ImgLevel = pd.DataFrame.from_dict({'id': image_id, 'PredictionString': PredString})\nImgLevel = pd.DataFrame(predarr, columns =['imageid', 'label','conf','xmin','ymin','xmax','ymax'])\nImgLevel.to_csv('ImagePredictions.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:44:06.022497Z","iopub.execute_input":"2021-08-04T13:44:06.022843Z","iopub.status.idle":"2021-08-04T13:44:08.999547Z","shell.execute_reply.started":"2021-08-04T13:44:06.022802Z","shell.execute_reply":"2021-08-04T13:44:08.998548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%rm ImagePredictions.csv","metadata":{"execution":{"iopub.status.busy":"2021-08-04T13:44:01.721916Z","iopub.execute_input":"2021-08-04T13:44:01.722351Z","iopub.status.idle":"2021-08-04T13:44:02.446001Z","shell.execute_reply.started":"2021-08-04T13:44:01.722316Z","shell.execute_reply":"2021-08-04T13:44:02.444761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cls_submission_df = cls_submission_df.append(ImgLevel).reset_index(drop=True)\ncls_submission_df.to_csv('/kaggle/working/submission.csv',index = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflow as tf\n# print(tf.__version__)\n# import torch\n# print(f\"Setup complete. Using torch {torch.__version__} ({torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else 'CPU'})\")\n\n# import gc\n# import glob\n# from tqdm import tqdm\n# from shutil import copyfile","metadata":{"execution":{"iopub.status.busy":"2021-08-03T05:43:20.530263Z","iopub.status.idle":"2021-08-03T05:43:20.530656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gpus = tf.config.list_physical_devices('GPU')\n# if gpus:\n#     try:\n#     # Currently, memory growth needs to be the same across GPUs\n#         for gpu in gpus:\n#             tf.config.experimental.set_memory_growth(gpu, True)\n#             logical_gpus = tf.config.experimental.list_logical_devices('GPU')\n#             print(len(gpus), \"Physical GPUs,\", len(logical_gpus), \"Logical GPUs\")\n#     except RuntimeError as e:\n#     # Memory growth must be set before GPUs have been initialized\n#         print(e)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:20.94349Z","iopub.execute_input":"2021-07-22T08:11:20.944192Z","iopub.status.idle":"2021-07-22T08:11:22.764736Z","shell.execute_reply.started":"2021-07-22T08:11:20.944145Z","shell.execute_reply":"2021-07-22T08:11:22.762897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\n# print(len(sub_df))\n# sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:22.768201Z","iopub.execute_input":"2021-07-22T08:11:22.768517Z","iopub.status.idle":"2021-07-22T08:11:22.797753Z","shell.execute_reply.started":"2021-07-22T08:11:22.768486Z","shell.execute_reply":"2021-07-22T08:11:22.796358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# study_df = sub_df.loc[sub_df.id.str.contains('_study')]\n# len(study_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:22.799746Z","iopub.execute_input":"2021-07-22T08:11:22.800237Z","iopub.status.idle":"2021-07-22T08:11:22.813758Z","shell.execute_reply.started":"2021-07-22T08:11:22.80019Z","shell.execute_reply":"2021-07-22T08:11:22.812232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_df = sub_df.loc[sub_df.id.str.contains('_image')]\n# len(image_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:22.816112Z","iopub.execute_input":"2021-07-22T08:11:22.816904Z","iopub.status.idle":"2021-07-22T08:11:22.82846Z","shell.execute_reply.started":"2021-07-22T08:11:22.816851Z","shell.execute_reply":"2021-07-22T08:11:22.826793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def read_xray(path, voi_lut = True, fix_monochrome = True):\n#     # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n#     dicom = pydicom.read_file(path)\n    \n#     # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n#     # \"human-friendly\" view\n#     if voi_lut:\n#         data = apply_voi_lut(dicom.pixel_array, dicom)\n#     else:\n#         data = dicom.pixel_array\n               \n#     # depending on this value, X-ray may look inverted - fix that:\n#     if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n#         data = np.amax(data) - data\n        \n#     data = data - np.min(data)\n#     data = data / np.max(data)\n#     data = (data * 255).astype(np.uint8)\n        \n#     return data\n\n# def resize_xray(array, size, keep_ratio=False, resample=Image.LANCZOS):\n#     # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n#     im = Image.fromarray(array)\n    \n#     if keep_ratio:\n#         im.thumbnail((size, size), resample)\n#     else:\n#         im = im.resize((size, size), resample)\n    \n#     return im","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:22.83082Z","iopub.execute_input":"2021-07-22T08:11:22.831416Z","iopub.status.idle":"2021-07-22T08:11:22.84423Z","shell.execute_reply.started":"2021-07-22T08:11:22.831368Z","shell.execute_reply":"2021-07-22T08:11:22.842809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TEST_PATH = f'/kaggle/tmp/test/'\n# IMG_SIZE = 512\n\n# def prepare_test_images():\n#     image_id = []\n#     dim0 = []\n#     dim1 = []\n\n#     os.makedirs(TEST_PATH, exist_ok=True)\n\n#     for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/test')):\n#         for file in filenames:\n#             # set keep_ratio=True to have original aspect ratio\n#             xray = read_xray(os.path.join(dirname, file))\n#             im = resize_xray(xray, size=IMG_SIZE)  \n#             im.save(os.path.join(TEST_PATH, file.replace('dcm', 'png')))\n\n#             image_id.append(file.replace('.dcm', ''))\n#             dim0.append(xray.shape[0])\n#             dim1.append(xray.shape[1])\n            \n#     return image_id, dim0, dim1","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:22.846035Z","iopub.execute_input":"2021-07-22T08:11:22.846562Z","iopub.status.idle":"2021-07-22T08:11:22.860096Z","shell.execute_reply.started":"2021-07-22T08:11:22.846516Z","shell.execute_reply":"2021-07-22T08:11:22.858771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_ids, dim0, dim1 = prepare_test_images()\n# print(f'Number of test images: {len(os.listdir(TEST_PATH))}')","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:11:22.862155Z","iopub.execute_input":"2021-07-22T08:11:22.862665Z","iopub.status.idle":"2021-07-22T08:21:24.591072Z","shell.execute_reply.started":"2021-07-22T08:11:22.862612Z","shell.execute_reply":"2021-07-22T08:21:24.586705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta_df = pd.DataFrame.from_dict({'image_id': image_ids, 'dim0': dim0, 'dim1': dim1})\n\n# # Associate image-level id with study-level ids.\n# # Note that a study-level might have more than one image-level ids.\n# for study_dir in os.listdir('../input/siim-covid19-detection/test'):\n#     for series in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}'):\n#         for image in os.listdir(f'../input/siim-covid19-detection/test/{study_dir}/{series}/'):\n#             image_id = image[:-4]\n#             meta_df.loc[meta_df['image_id'] == image_id, 'study_id'] = study_dir\n        \n# meta_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:21:24.592863Z","iopub.execute_input":"2021-07-22T08:21:24.593359Z","iopub.status.idle":"2021-07-22T08:21:27.417353Z","shell.execute_reply.started":"2021-07-22T08:21:24.593313Z","shell.execute_reply":"2021-07-22T08:21:27.416051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cp /kaggle/input/yolomodels-obj/yolov5L512_1607.pt /kaggle/working\n# %cp /kaggle/input/yolomodels-obj/yolov5m6_512.pt /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:21:27.418994Z","iopub.execute_input":"2021-07-22T08:21:27.419454Z","iopub.status.idle":"2021-07-22T08:21:33.223583Z","shell.execute_reply.started":"2021-07-22T08:21:27.419409Z","shell.execute_reply":"2021-07-22T08:21:33.222165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !python /kaggle/input/siimcovidyolov5l/yolov5/detect.py --weights /kaggle/working/yolov5L512_1607.pt /kaggle/working/yolov5m6_512.pt \\\n#                                       --source {TEST_PATH} \\\n#                                       --img 512 \\\n#                                       --conf 0.22 \\\n#                                       --iou-thres 0.5 \\\n#                                       --max-det 10 \\\n#                                       --save-txt \\\n#                                       --save-conf \\\n#                                       --augment","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:21:33.227743Z","iopub.execute_input":"2021-07-22T08:21:33.228094Z","iopub.status.idle":"2021-07-22T08:24:31.166607Z","shell.execute_reply.started":"2021-07-22T08:21:33.228059Z","shell.execute_reply":"2021-07-22T08:24:31.165306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PRED_PATH = 'runs/detect/exp/labels'\n# prediction_files = os.listdir(PRED_PATH)\n# print(f'Number of opacity predicted by YOLOv5: {len(prediction_files)}')","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:24:31.170752Z","iopub.execute_input":"2021-07-22T08:24:31.171123Z","iopub.status.idle":"2021-07-22T08:24:31.181945Z","shell.execute_reply.started":"2021-07-22T08:24:31.171087Z","shell.execute_reply":"2021-07-22T08:24:31.180127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def correct_bbox_format(bboxes):\n#     correct_bboxes = []\n#     for b in bboxes:\n#         xc, yc = int(np.round(b[0]*IMG_SIZE)), int(np.round(b[1]*IMG_SIZE))\n#         w, h = int(np.round(b[2]*IMG_SIZE)), int(np.round(b[3]*IMG_SIZE))\n\n#         xmin = xc - int(np.round(w/2))\n#         ymin = yc - int(np.round(h/2))\n#         xmax = xc + int(np.round(w/2))\n#         ymax = yc + int(np.round(h/2))\n        \n#         correct_bboxes.append([xmin, ymin, xmax, ymax])\n        \n#     return correct_bboxes\n\n# def scale_bboxes_to_original(row, bboxes):\n#     # Get scaling factor\n#     scale_x = IMG_SIZE/row.dim1\n#     scale_y = IMG_SIZE/row.dim0\n    \n#     scaled_bboxes = []\n#     for bbox in bboxes:\n#         xmin, ymin, xmax, ymax = bbox\n        \n#         xmin = int(np.round(xmin/scale_x))\n#         ymin = int(np.round(ymin/scale_y))\n#         xmax = int(np.round(xmax/scale_x))\n#         ymax = int(np.round(ymax/scale_y))\n        \n#         scaled_bboxes.append([xmin, ymin, xmax, ymax])\n        \n#     return scaled_bboxes\n\n# # Read the txt file generated by YOLOv5 during inference and extract \n# # confidence and bounding box coordinates.\n# def get_conf_bboxes(file_path):\n#     confidence = []\n#     bboxes = []\n#     with open(file_path, 'r') as file:\n#         for line in file:\n#             preds = line.strip('\\n').split(' ')\n#             preds = list(map(float, preds))\n#             confidence.append(preds[-1])\n#             bboxes.append(preds[1:-1])\n#     return confidence, bboxes","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:24:31.18411Z","iopub.execute_input":"2021-07-22T08:24:31.184616Z","iopub.status.idle":"2021-07-22T08:24:31.200883Z","shell.execute_reply.started":"2021-07-22T08:24:31.18457Z","shell.execute_reply":"2021-07-22T08:24:31.199195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_pred_strings = []\n# ctr = 0\n# for i in tqdm(range(len(image_df))):\n#     row = meta_df.loc[i]\n#     id_name = row.image_id\n    \n#     if f'{id_name}.txt' in prediction_files:\n#         # opacity label\n#         confidence, bboxes = get_conf_bboxes(f'{PRED_PATH}/{id_name}.txt')\n#         bboxes = correct_bbox_format(bboxes)\n#         ori_bboxes = scale_bboxes_to_original(row, bboxes)\n        \n#         pred_string = ''\n#         for j, conf in enumerate(confidence):\n#             pred_string += f'opacity {conf} ' + ' '.join(map(str, ori_bboxes[j])) + ' '\n        \n#         row = mean_predictions_df.loc[mean_predictions_df['study_id'] == row.study_id]\n#         neg = row.Negative.item()\n#         typ = row.Typical.item()\n#         ind = row.Indeterminate.item()\n#         atp = row.Atypical.item()\n#         output_class = np.argmax(np.array([neg,typ,ind,atp]))\n#         if output_class == 0 and neg > 0.7:\n#             ctr+=1\n#             image_pred_strings.append(\"none 1 0 0 1 1\")\n#         else:\n#             image_pred_strings.append(pred_string[:-1])\n#     else:\n#         image_pred_strings.append(\"none 1 0 0 1 1\")\n# print('Number of images that were detected as opacity but are forced to none on the basis of classification output are :' + str(ctr))","metadata":{"execution":{"iopub.status.busy":"2021-07-22T09:25:24.014807Z","iopub.execute_input":"2021-07-22T09:25:24.015289Z","iopub.status.idle":"2021-07-22T09:25:25.404289Z","shell.execute_reply.started":"2021-07-22T09:25:24.015255Z","shell.execute_reply":"2021-07-22T09:25:25.402781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta_df['PredictionString'] = image_pred_strings\n# image_df = meta_df[['study_id','image_id', 'PredictionString']]\n# # image_df.insert(0, 'id', image_df.apply(lambda row: row.image_id+'_image', axis=1))\n# # image_df = image_df.drop('image_id', axis=1)\n# image_df.head(20)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:24:31.834683Z","iopub.execute_input":"2021-07-22T08:24:31.835035Z","iopub.status.idle":"2021-07-22T08:24:31.855681Z","shell.execute_reply.started":"2021-07-22T08:24:31.835002Z","shell.execute_reply":"2021-07-22T08:24:31.854428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_df.insert(0, 'id', image_df.apply(lambda row: row.image_id+'_image', axis=1))","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:24:31.857463Z","iopub.execute_input":"2021-07-22T08:24:31.857917Z","iopub.status.idle":"2021-07-22T08:24:31.902856Z","shell.execute_reply.started":"2021-07-22T08:24:31.857869Z","shell.execute_reply":"2021-07-22T08:24:31.901649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_df_new = image_df[['id','PredictionString']]","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:24:31.90461Z","iopub.execute_input":"2021-07-22T08:24:31.905074Z","iopub.status.idle":"2021-07-22T08:24:31.912427Z","shell.execute_reply.started":"2021-07-22T08:24:31.905027Z","shell.execute_reply":"2021-07-22T08:24:31.910848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cls_submission_df = cls_submission_df.append(image_df_new).reset_index(drop=True)\n# cls_submission_df.to_csv('/kaggle/working/submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2021-07-22T08:24:31.914479Z","iopub.execute_input":"2021-07-22T08:24:31.915036Z","iopub.status.idle":"2021-07-22T08:24:31.945199Z","shell.execute_reply.started":"2021-07-22T08:24:31.914989Z","shell.execute_reply":"2021-07-22T08:24:31.944069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%rm -rf runs\n%rm -rf test_384x384\n%rm -rf test_512x512\n%rm -rf ImagePredictions.csv\n%rm -rf ImageLevelPredictions.csv","metadata":{},"execution_count":null,"outputs":[]}]}