{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%load_ext autoreload\n%autoreload 2\n\n!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -y --offline\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -y --offline\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -y --offline\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -y --offline\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -y --offline\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -y --offline\n\n!pip install '/kaggle/input/kerasapplications' --no-deps\n!pip install /kaggle/input/efficientnet-keras-source-code/ -q --no-deps\n%cd /kaggle/working/","metadata":{"papermill":{"duration":589.21781,"end_time":"2021-07-17T19:03:09.933611","exception":false,"start_time":"2021-07-17T18:53:20.715801","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-09T13:21:55.993896Z","iopub.execute_input":"2021-08-09T13:21:55.994301Z","iopub.status.idle":"2021-08-09T13:23:55.404951Z","shell.execute_reply.started":"2021-08-09T13:21:55.994211Z","shell.execute_reply":"2021-08-09T13:23:55.403651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install '/kaggle/input/kerasapplications' --no-deps\n!pip install '/kaggle/input/efficientnet-keras-source-code' --no-deps\n!pip install '/kaggle/input/effdet-latestvinbigdata-wbf-fused/ensemble_boxes-1.0.4-py3-none-any.whl' --no-deps\n\n## MMDetection compatible torch installation\n!pip install '/kaggle/input/pytorch-170-cuda-toolkit-110221/torch-1.7.0+cu110-cp37-cp37m-linux_x86_64.whl' --no-deps\n!pip install '/kaggle/input/pytorch-170-cuda-toolkit-110221/torchvision-0.8.1+cu110-cp37-cp37m-linux_x86_64.whl' --no-deps\n!pip install '/kaggle/input/pytorch-170-cuda-toolkit-110221/torchaudio-0.7.0-cp37-cp37m-linux_x86_64.whl' --no-deps\n\n## Compatible Cuda Toolkit installation\n!mkdir -p /kaggle/tmp && cp /kaggle/input/pytorch-170-cuda-toolkit-110221/cudatoolkit-11.0.221-h6bb024c_0 /kaggle/tmp/cudatoolkit-11.0.221-h6bb024c_0.tar.bz2 && conda install /kaggle/tmp/cudatoolkit-11.0.221-h6bb024c_0.tar.bz2 -y --offline","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:23:55.406897Z","iopub.execute_input":"2021-08-09T13:23:55.407278Z","iopub.status.idle":"2021-08-09T13:29:45.22571Z","shell.execute_reply.started":"2021-08-09T13:23:55.407236Z","shell.execute_reply":"2021-08-09T13:29:45.224729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## MMDetection Offline Installation\n!pip install '/kaggle/input/mmdetectionv2140/addict-2.4.0-py3-none-any.whl' --no-deps\n!pip install '/kaggle/input/mmdetectionv2140/yapf-0.31.0-py2.py3-none-any.whl' --no-deps\n!pip install '/kaggle/input/mmdetectionv2140/terminal-0.4.0-py3-none-any.whl' --no-deps\n!pip install '/kaggle/input/mmdetectionv2140/terminaltables-3.1.0-py3-none-any.whl' --no-deps\n!pip install '/kaggle/input/mmdetectionv2140/mmcv_full-1_3_8-cu110-torch1_7_0/mmcv_full-1.3.8-cp37-cp37m-manylinux1_x86_64.whl' --no-deps\n!pip install '/kaggle/input/mmdetectionv2140/pycocotools-2.0.2/pycocotools-2.0.2' --no-deps\n!pip install '/kaggle/input/mmdetectionv2140/mmpycocotools-12.0.3/mmpycocotools-12.0.3' --no-deps\n\n!cp -r /kaggle/input/mmdetectionv2140/mmdetection-2.14.0 /kaggle/working/\n!mv /kaggle/working/mmdetection-2.14.0 /kaggle/working/mmdetection\n%cd /kaggle/working/mmdetection\n!pip install -e . --no-deps\n%cd /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:29:45.2294Z","iopub.execute_input":"2021-08-09T13:29:45.229665Z","iopub.status.idle":"2021-08-09T13:33:11.946176Z","shell.execute_reply.started":"2021-08-09T13:29:45.229637Z","shell.execute_reply":"2021-08-09T13:33:11.945328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/working/mmdetection')\n\nimport os\nimport shutil\nimport gc\nfrom glob import glob\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\nimport efficientnet.tfkeras as efn\nimport tensorflow as tf\nimport random\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.104274,"end_time":"2021-07-17T19:03:10.131834","exception":false,"start_time":"2021-07-17T19:03:10.02756","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-09T13:33:11.94808Z","iopub.execute_input":"2021-08-09T13:33:11.948462Z","iopub.status.idle":"2021-08-09T13:33:17.017665Z","shell.execute_reply.started":"2021-08-09T13:33:11.94842Z","shell.execute_reply":"2021-08-09T13:33:17.016663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #00857e; font-family: Segoe UI; font-size: 1.8em; font-weight: 300;\">Create Study and Image Level Dataframes</span>","metadata":{"papermill":{"duration":0.092535,"end_time":"2021-07-17T19:03:10.317337","exception":false,"start_time":"2021-07-17T19:03:10.224802","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\n\n# Form study and image dataframes\nsub_df['level'] = sub_df.id.map(lambda idx: idx[-5:])\nstudy_df = sub_df[sub_df.level=='study'].rename({'id':'study_id'}, axis=1)\nimage_df = sub_df[sub_df.level=='image'].rename({'id':'image_id'}, axis=1)\n\ndcm_path = glob('/kaggle/input/siim-covid19-detection/test/**/*dcm', recursive=True)\ntest_meta = pd.DataFrame({'dcm_path':dcm_path})\ntest_meta['image_id'] = test_meta.dcm_path.map(lambda x: x.split('/')[-1].replace('.dcm', '')+'_image')\ntest_meta['study_id'] = test_meta.dcm_path.map(lambda x: x.split('/')[-3].replace('.dcm', '')+'_study')\n\nstudy_df = study_df.merge(test_meta, on='study_id', how='left')\nimage_df = image_df.merge(test_meta, on='image_id', how='left')\n\n# Remove duplicates study_ids from study_df\nstudy_df.drop_duplicates(subset=\"study_id\",keep='first', inplace=True)","metadata":{"papermill":{"duration":5.695531,"end_time":"2021-07-17T19:03:16.105354","exception":false,"start_time":"2021-07-17T19:03:10.409823","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-09T13:33:17.019133Z","iopub.execute_input":"2021-08-09T13:33:17.019507Z","iopub.status.idle":"2021-08-09T13:33:22.247085Z","shell.execute_reply.started":"2021-08-09T13:33:17.019466Z","shell.execute_reply":"2021-08-09T13:33:22.246086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #00857e; font-family: Segoe UI; font-size: 1.8em; font-weight: 300;\">Fast or Full Predictions</span>\n\nIn case of non-competetion submission commits, we run the notebook with just two images each for image level and study level inference from the public test data.","metadata":{"papermill":{"duration":0.154867,"end_time":"2021-07-17T19:03:16.415952","exception":false,"start_time":"2021-07-17T19:03:16.261085","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fast_sub = False\n\nif sub_df.shape[0] == 2477:\n    fast_sub = True\n    study_df = study_df.sample(2)\n    image_df = image_df.sample(2)\n    \n    print(\"\\nstudy_df\")\n    display(study_df.head(2))\n    print(\"\\nimage_df\")\n    display(image_df.head(2))\n    print(\"\\ntest_meta\")\n    display(test_meta.head(2))","metadata":{"papermill":{"duration":0.28084,"end_time":"2021-07-17T19:03:16.86864","exception":false,"start_time":"2021-07-17T19:03:16.5878","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-09T13:33:22.2485Z","iopub.execute_input":"2021-08-09T13:33:22.248913Z","iopub.status.idle":"2021-08-09T13:33:22.360847Z","shell.execute_reply.started":"2021-08-09T13:33:22.248864Z","shell.execute_reply":"2021-08-09T13:33:22.359591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fast_sub = False","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:33:22.362342Z","iopub.execute_input":"2021-08-09T13:33:22.362768Z","iopub.status.idle":"2021-08-09T13:33:22.429416Z","shell.execute_reply.started":"2021-08-09T13:33:22.362704Z","shell.execute_reply":"2021-08-09T13:33:22.428401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nSTUDY_DIMS = (768, 768)\nIMAGE_DIMS = (512, 512)\n\nstudy_dir = f'/kaggle/tmp/test/study/'\nos.makedirs(study_dir, exist_ok=True)\n\nimage_dir = f'/kaggle/tmp/test/image/'\nos.makedirs(image_dir, exist_ok=True)\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    return im\n\nfor index, row in tqdm(study_df[['study_id', 'dcm_path']].iterrows(), total=study_df.shape[0]):\n    # set keep_ratio=True to have original aspect ratio\n    xray = read_xray(row['dcm_path'])\n    im = resize(xray, size=STUDY_DIMS[0])\n    im.save(os.path.join(study_dir, row['study_id']+'.png'))\n\n# image_df['dim0'] = -1\n# image_df['dim1'] = -1\n\n# for index, row in tqdm(image_df[['image_id', 'dcm_path', 'dim0', 'dim1']].iterrows(), total=image_df.shape[0]):\n#     # set keep_ratio=True to have original aspect ratio\n#     xray = read_xray(row['dcm_path'])\n#     im = resize(xray, size=IMAGE_DIMS[0])  \n#     im.save(os.path.join(image_dir, row['image_id']+'.png'))\n#     image_df.loc[image_df.image_id==row.image_id, 'dim0'] = xray.shape[0]\n#     image_df.loc[image_df.image_id==row.image_id, 'dim1'] = xray.shape[1]","metadata":{"papermill":{"duration":4.178244,"end_time":"2021-07-17T19:03:21.225656","exception":false,"start_time":"2021-07-17T19:03:17.047412","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-09T13:33:22.432947Z","iopub.execute_input":"2021-08-09T13:33:22.433612Z","iopub.status.idle":"2021-08-09T13:33:24.186729Z","shell.execute_reply.started":"2021-08-09T13:33:22.433567Z","shell.execute_reply":"2021-08-09T13:33:24.185807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df['image_path'] = study_dir+study_df['study_id']+'.png'\n#image_df['image_path'] = image_dir+image_df['image_id']+'.png'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:33:24.188777Z","iopub.execute_input":"2021-08-09T13:33:24.189159Z","iopub.status.idle":"2021-08-09T13:33:24.263634Z","shell.execute_reply.started":"2021-08-09T13:33:24.189121Z","shell.execute_reply":"2021-08-09T13:33:24.262816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #00857e; font-family: Segoe UI; font-size: 1.8em; font-weight: 300;\">Custom Wrapper for Loading TFHub Model trained in TPU</span>\n\nSince the EffNetV2 Classifier models were trained on a TPU with the `tfhub.KerasLayer` formed with the handle argument as a GCS path, while loading the saved model for inference, the method tries to download the pre-trained weights from the definition of the layer from training i.e a GCS path.\n\nSince, inference notebooks don't have GCS and internet access, it is not possible to load the model without the pretrained weights explicitly loaded from the local directory.\n\nIf the models were trained on a GPU, we can use the cache location method to load the pre-trained weights by storing them in a cache folder with the hashed key of the model location, as the folder name. I tried this method here but, it doesn't seem to work as the model was trained with a GCS path defined in the `tfhub.KerasLayer` and the method kept on hitting the GCS path rather than loading the weights from the cache location.\n\nThe only solution was to create a wrapper class to correct the handle argument to load the right pretrained weights explicitly from the local directory.","metadata":{"papermill":{"duration":0.096404,"end_time":"2021-07-17T19:03:21.416262","exception":false,"start_time":"2021-07-17T19:03:21.319858","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as tfhub\n\nMODEL_ARCH = 'efficientnetv2-l-21k-ft1k'\n# Get the TensorFlow Hub model URL\nhub_type = 'feature_vector' # ['classification', 'feature_vector']\nMODEL_ARCH_PATH = f'/kaggle/input/efficientnetv2-tfhub-weight-files/tfhub_models/{MODEL_ARCH}/{hub_type}'\n\n# Custom wrapper class to load the right pretrained weights explicitly from the local directory\nclass KerasLayerWrapper(tfhub.KerasLayer):\n    def __init__(self, handle, **kwargs):\n        handle = tfhub.KerasLayer(tfhub.load(MODEL_ARCH_PATH))\n        super().__init__(handle, **kwargs)","metadata":{"papermill":{"duration":4.259738,"end_time":"2021-07-17T19:03:25.768808","exception":false,"start_time":"2021-07-17T19:03:21.50907","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-09T13:33:24.265017Z","iopub.execute_input":"2021-08-09T13:33:24.265391Z","iopub.status.idle":"2021-08-09T13:33:24.610229Z","shell.execute_reply.started":"2021-08-09T13:33:24.265362Z","shell.execute_reply":"2021-08-09T13:33:24.609246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #00857e; font-family: Segoe UI; font-size: 1.8em; font-weight: 300;\">Predict Study Level</span>","metadata":{"papermill":{"duration":0.087465,"end_time":"2021-07-17T19:03:25.943841","exception":false,"start_time":"2021-07-17T19:03:25.856376","status":"completed"},"tags":[]}},{"cell_type":"code","source":"MODEL_PATH = '/kaggle/input/siim-effnetv2-keras-study-train-tpu-cv0-805'\nMODEL_PATH = '/kaggle/input/siims-c19-efnb7-pl-weights-study-only' # With pseudo labels and efnb2 , folder name is wrong\n\ntest_paths = study_df.image_path.tolist()\nBATCH_SIZE = 16\n\ndef build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n\n    def decode_with_labels(path, label):\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n\n    def augment_with_labels(img, label):\n        return augment(img), label\n\n    return augment_with_labels if with_labels else augment\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n\n    return dset\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:33:24.611568Z","iopub.execute_input":"2021-08-09T13:33:24.611959Z","iopub.status.idle":"2021-08-09T13:33:24.687842Z","shell.execute_reply.started":"2021-08-09T13:33:24.611922Z","shell.execute_reply":"2021-08-09T13:33:24.686895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# strategy = auto_select_accelerator()\n# BATCH_SIZE = strategy.num_replicas_in_sync * 16\n\nlabel_cols = ['negative', 'typical', 'indeterminate', 'atypical']\nstudy_df[label_cols] = 0\n\ntest_decoder = build_decoder(with_labels=False,\n                             target_size=(STUDY_DIMS[0],\n                                          STUDY_DIMS[0]), ext='png')\ntest_dataset = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)\n\nwith tf.device('/device:GPU:0'):\n    models = []\n    models0 = tf.keras.models.load_model(f'{MODEL_PATH}/model0.h5',\n                                         custom_objects={'KerasLayer': KerasLayerWrapper})\n    models1 = tf.keras.models.load_model(f'{MODEL_PATH}/model1.h5',\n                                         custom_objects={'KerasLayer': KerasLayerWrapper})\n    models2 = tf.keras.models.load_model(f'{MODEL_PATH}/model2.h5',\n                                         custom_objects={'KerasLayer': KerasLayerWrapper})\n    models3 = tf.keras.models.load_model(f'{MODEL_PATH}/model3.h5',\n                                         custom_objects={'KerasLayer': KerasLayerWrapper})\n    models4 = tf.keras.models.load_model(f'{MODEL_PATH}/model4.h5',\n                                         custom_objects={'KerasLayer': KerasLayerWrapper})\n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    models.append(models4)\n\n# study_df[label_cols] = sum([model.predict(test_dataset, verbose=1) for model in models]) / len(models)\n# study_df['PredictionString'] = study_df[label_cols].apply(lambda row: f'negative {row.negative} 0 0 1 1 typical {row.typical} 0 0 1 1 indeterminate {row.indeterminate} 0 0 1 1 atypical {row.atypical} 0 0 1 1', axis=1)\n\n# del models\n# del models0, models1, models2, models3, models4\n# del test_dataset, test_decoder\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:33:24.689328Z","iopub.execute_input":"2021-08-09T13:33:24.689717Z","iopub.status.idle":"2021-08-09T13:36:47.957096Z","shell.execute_reply.started":"2021-08-09T13:33:24.689678Z","shell.execute_reply":"2021-08-09T13:36:47.956174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = {\n    0: 1,\n    1: 1,\n    2: 1,\n    3: 1,\n    4: 1,\n}\n\nweights_sum = sum(weights.values())\nweights = {k: v/weights_sum for k, v in weights.items()}\n\npredictions1 = [model.predict(test_dataset, verbose=1) for model in models]\nfor i, pred in enumerate(predictions1):\n    predictions1[i] = weights[i] * pred","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:36:47.958381Z","iopub.execute_input":"2021-08-09T13:36:47.958734Z","iopub.status.idle":"2021-08-09T13:37:12.774261Z","shell.execute_reply.started":"2021-08-09T13:36:47.958699Z","shell.execute_reply":"2021-08-09T13:37:12.773272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dataset, test_decoder\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:37:12.775905Z","iopub.execute_input":"2021-08-09T13:37:12.776301Z","iopub.status.idle":"2021-08-09T13:37:15.500788Z","shell.execute_reply.started":"2021-08-09T13:37:12.776258Z","shell.execute_reply":"2021-08-09T13:37:15.499787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_decoder = build_decoder(with_labels=False,\n                             target_size=(600,\n                                          600), ext='png')\ntest_dataset = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:37:15.502297Z","iopub.execute_input":"2021-08-09T13:37:15.502639Z","iopub.status.idle":"2021-08-09T13:37:15.5804Z","shell.execute_reply.started":"2021-08-09T13:37:15.502609Z","shell.execute_reply":"2021-08-09T13:37:15.579562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:37:15.58171Z","iopub.execute_input":"2021-08-09T13:37:15.582273Z","iopub.status.idle":"2021-08-09T13:37:15.645731Z","shell.execute_reply.started":"2021-08-09T13:37:15.582229Z","shell.execute_reply":"2021-08-09T13:37:15.644813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/device:GPU:0'):\n    models = []\n    models5 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainstudy-classification-weights/model0.h5',custom_objects={'KerasLayer': KerasLayerWrapper}\n    )\n    models6 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainstudy-classification-weights/model1.h5',custom_objects={'KerasLayer': KerasLayerWrapper}\n    )\n    models7 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainstudy-classification-weights/model2.h5',custom_objects={'KerasLayer': KerasLayerWrapper}\n    )\n    models8 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainstudy-classification-weights/model3.h5',custom_objects={'KerasLayer': KerasLayerWrapper}\n    )\n    models9 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainstudy-classification-weights/model4.h5',custom_objects={'KerasLayer': KerasLayerWrapper}\n    )\n    models.append(models5)\n    models.append(models6)\n    models.append(models7)\n    models.append(models8)\n    models.append(models9)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:37:15.647143Z","iopub.execute_input":"2021-08-09T13:37:15.647569Z","iopub.status.idle":"2021-08-09T13:38:57.947142Z","shell.execute_reply.started":"2021-08-09T13:37:15.64753Z","shell.execute_reply":"2021-08-09T13:38:57.946093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = {\n    0: 1,\n    1: 1,\n    2: 1,\n    3: 1,\n    4: 1,\n}\n\nweights_sum = sum(weights.values())\nweights = {k: v/weights_sum for k, v in weights.items()}\n\npredictions2 = [model.predict(test_dataset, verbose=1) for model in models]\nfor i, pred in enumerate(predictions2):\n    predictions2[i] = weights[i] * pred","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:38:57.948695Z","iopub.execute_input":"2021-08-09T13:38:57.949112Z","iopub.status.idle":"2021-08-09T13:39:20.44365Z","shell.execute_reply.started":"2021-08-09T13:38:57.949071Z","shell.execute_reply":"2021-08-09T13:39:20.442768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions1 = list(np.array(predictions1)*.7)\npredictions2 = list(np.array(predictions2)*.3)\npredictions = np.concatenate((predictions1, predictions2), axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:20.445097Z","iopub.execute_input":"2021-08-09T13:39:20.445475Z","iopub.status.idle":"2021-08-09T13:39:20.509008Z","shell.execute_reply.started":"2021-08-09T13:39:20.445435Z","shell.execute_reply":"2021-08-09T13:39:20.508147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df[label_cols] = sum(predictions)\nstudy_df['PredictionString'] = study_df[label_cols].apply(lambda row: f'negative {row.negative} 0 0 1 1 typical {row.typical} 0 0 1 1 indeterminate {row.indeterminate} 0 0 1 1 atypical {row.atypical} 0 0 1 1', axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:20.510409Z","iopub.execute_input":"2021-08-09T13:39:20.510804Z","iopub.status.idle":"2021-08-09T13:39:20.585193Z","shell.execute_reply.started":"2021-08-09T13:39:20.510767Z","shell.execute_reply":"2021-08-09T13:39:20.584207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del models\ndel models0, models1, models2, models3, models4, models5, models6, models7, models8, models9\ndel test_dataset, test_decoder\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:20.586545Z","iopub.execute_input":"2021-08-09T13:39:20.587166Z","iopub.status.idle":"2021-08-09T13:39:35.568781Z","shell.execute_reply.started":"2021-08-09T13:39:20.587127Z","shell.execute_reply":"2021-08-09T13:39:35.5678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df.rename(columns = {'study_id':'id'}, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:35.576821Z","iopub.execute_input":"2021-08-09T13:39:35.577115Z","iopub.status.idle":"2021-08-09T13:39:35.668796Z","shell.execute_reply.started":"2021-08-09T13:39:35.577086Z","shell.execute_reply":"2021-08-09T13:39:35.667789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:35.674877Z","iopub.execute_input":"2021-08-09T13:39:35.675166Z","iopub.status.idle":"2021-08-09T13:39:35.747063Z","shell.execute_reply.started":"2021-08-09T13:39:35.67514Z","shell.execute_reply":"2021-08-09T13:39:35.745806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 Class","metadata":{}},{"cell_type":"code","source":"import efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:35.751597Z","iopub.execute_input":"2021-08-09T13:39:35.751985Z","iopub.status.idle":"2021-08-09T13:39:35.813641Z","shell.execute_reply.started":"2021-08-09T13:39:35.751956Z","shell.execute_reply":"2021-08-09T13:39:35.812533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split = 'test'\nimage_id = []\ndim0 = []\ndim1 = []\nsplits = []\n\nsave_dir = f'/kaggle/tmp/{split}/image/'\nos.makedirs(save_dir, exist_ok=True)\n\nif fast_sub:\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'/kaggle/input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=512)  \n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)\n\nmeta = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:39:35.815463Z","iopub.execute_input":"2021-08-09T13:39:35.816002Z","iopub.status.idle":"2021-08-09T13:47:39.237548Z","shell.execute_reply.started":"2021-08-09T13:39:35.81596Z","shell.execute_reply":"2021-08-09T13:47:39.236785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n\n    return strategy\n","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:39.238823Z","iopub.execute_input":"2021-08-09T13:47:39.239146Z","iopub.status.idle":"2021-08-09T13:47:39.447417Z","shell.execute_reply.started":"2021-08-09T13:47:39.23912Z","shell.execute_reply":"2021-08-09T13:47:39.446468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512)\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\n\nif fast_sub:\n    fast_df = pd.DataFrame(([['0838c0a2c424_study', 'negative 1 0 0 1 1'], \n                         ['677ad87244f5_study', 'negative 1 0 0 1 1'], \n                         ['65761e66de9f_image', 'none 1 0 0 1 1'], \n                         ['51759b5579bc_image', 'none 1 0 0 1 1']]), \n                       columns=['id', 'PredictionString'])\n    df = fast_df.copy()\nelse:\n    df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\n    \nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\n\ndf['id_last_str'] = id_laststr_list\n\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:39.448875Z","iopub.execute_input":"2021-08-09T13:47:39.449249Z","iopub.status.idle":"2021-08-09T13:47:39.563418Z","shell.execute_reply.started":"2021-08-09T13:47:39.449212Z","shell.execute_reply":"2021-08-09T13:47:39.562487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n#     def decode(path):\n#         file_bytes = tf.io.read_file(path)\n#         if ext == 'png':\n#             img = tf.image.decode_png(file_bytes, channels=3)\n#         elif ext in ['jpg', 'jpeg']:\n#             img = tf.image.decode_jpeg(file_bytes, channels=3)\n#         else:\n#             raise ValueError(\"Image extension not supported\")\n\n#         img = tf.cast(img, tf.float32) / 255.0\n#         img = tf.image.resize(img, target_size)\n\n#         return img\n\n#     def decode_with_labels(path, label):\n#         return decode(path), label\n\n#     return decode_with_labels if with_labels else decode","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:39.564726Z","iopub.execute_input":"2021-08-09T13:47:39.565081Z","iopub.status.idle":"2021-08-09T13:47:39.625436Z","shell.execute_reply.started":"2021-08-09T13:47:39.565044Z","shell.execute_reply":"2021-08-09T13:47:39.624488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\n\nsub_df = sub_df[study_len:]\ntest_paths = f'/kaggle/tmp/test/image/' + sub_df['id'] +'.png'\nsub_df['none'] = 0\n\nlabel_cols = sub_df.columns[2]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:39.628725Z","iopub.execute_input":"2021-08-09T13:47:39.629048Z","iopub.status.idle":"2021-08-09T13:47:39.704707Z","shell.execute_reply.started":"2021-08-09T13:47:39.62902Z","shell.execute_reply":"2021-08-09T13:47:39.703903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[8], IMSIZE[8]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:39.706265Z","iopub.execute_input":"2021-08-09T13:47:39.706663Z","iopub.status.idle":"2021-08-09T13:47:39.784634Z","shell.execute_reply.started":"2021-08-09T13:47:39.706614Z","shell.execute_reply":"2021-08-09T13:47:39.783843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/siimc19efnb7trainimage-classification-weights/","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:39.786056Z","iopub.execute_input":"2021-08-09T13:47:39.786446Z","iopub.status.idle":"2021-08-09T13:47:40.734029Z","shell.execute_reply.started":"2021-08-09T13:47:39.786407Z","shell.execute_reply":"2021-08-09T13:47:40.732918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    models = []\n    \n    models0 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainimage-classification-weights/model0.h5'\n    )    \n    models1 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainimage-classification-weights/model1.h5'\n    )\n    models2 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainimage-classification-weights/model2.h5'\n    )\n    models3 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainimage-classification-weights/model3.h5'\n    )\n    models4 = tf.keras.models.load_model(\n        '../input/siimc19efnb7trainimage-classification-weights/model4.h5'\n    )\n    models5 = tf.keras.models.load_model(\n        '../input/b77trainyukita/model0.h5'\n    )\n    models6 = tf.keras.models.load_model(\n        '../input/b77trainyukita/model1.h5'\n    )\n    models7 = tf.keras.models.load_model(\n        '../input/b77trainyukita/model2.h5'\n    )\n    models8 = tf.keras.models.load_model(\n        '../input/b77trainyukita/model3.h5'\n    )\n    models9 = tf.keras.models.load_model(\n        '../input/b77trainyukita/model4.h5'\n    )\n    models10 = tf.keras.models.load_model(\n        '../input/restnet152modelsyukit/model3.h5'\n    )\n    models11 = tf.keras.models.load_model(\n        '../input/restnet152modelsyukit/model2.h5'\n    )\n    models12 = tf.keras.models.load_model(\n        '../input/restnet152modelsyukit/model1.h5'\n    )\n    models13 = tf.keras.models.load_model(\n        '../input/restnet152modelsyukit/model4.h5'\n    )\n    models14 = tf.keras.models.load_model(\n        '../input/restnet152modelsyukit/model0.h5'\n    )\n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    models.append(models4)\n    models.append(models5)\n    models.append(models6)\n    models.append(models7)\n    models.append(models8)\n    models.append(models9)\n    models.append(models10)\n    models.append(models11)\n    models.append(models12)\n    models.append(models13)\n    models.append(models14)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:47:40.735728Z","iopub.execute_input":"2021-08-09T13:47:40.736152Z","iopub.status.idle":"2021-08-09T13:52:03.651565Z","shell.execute_reply.started":"2021-08-09T13:47:40.736117Z","shell.execute_reply":"2021-08-09T13:52:03.650553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/tmp/test/image/","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:52:03.653329Z","iopub.execute_input":"2021-08-09T13:52:03.653731Z","iopub.status.idle":"2021-08-09T13:52:04.619952Z","shell.execute_reply.started":"2021-08-09T13:52:03.653686Z","shell.execute_reply":"2021-08-09T13:52:04.618802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights = {\n    0: 3.25,\n    1: 2,\n    2: 2,\n    3: 2,\n    4: 2,\n    5: 2,\n    6: 1,\n    7: 1,\n    8: 1,\n    9: 2,\n    10: 0,\n    11: 1.75,\n    12: 0,\n    13: 0,\n    14: 1.75\n    \n}\n\nweights_sum = sum(weights.values())\nweights = {k: v/weights_sum for k, v in weights.items()}\n\npredictions = [model.predict(dtest, verbose=1) for model in models]\nfor i, pred in enumerate(predictions):\n    predictions[i] = weights[i] * pred\n    \nsub_df[label_cols] = sum(predictions)\n\ndf_2class = sub_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T13:52:04.621764Z","iopub.execute_input":"2021-08-09T13:52:04.622227Z","iopub.status.idle":"2021-08-09T14:01:11.899142Z","shell.execute_reply.started":"2021-08-09T13:52:04.622185Z","shell.execute_reply":"2021-08-09T14:01:11.898217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_2class","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:11.900667Z","iopub.execute_input":"2021-08-09T14:01:11.901074Z","iopub.status.idle":"2021-08-09T14:01:11.984967Z","shell.execute_reply.started":"2021-08-09T14:01:11.901033Z","shell.execute_reply":"2021-08-09T14:01:11.983924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel models\ndel models0, models1, models2, models3, models4, models5, models6, models7, models8, models9, models10, models11, models12, models13, models14\ngc.collect","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:11.986386Z","iopub.execute_input":"2021-08-09T14:01:11.98698Z","iopub.status.idle":"2021-08-09T14:01:12.058694Z","shell.execute_reply.started":"2021-08-09T14:01:11.986939Z","shell.execute_reply":"2021-08-09T14:01:12.057619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numba import cuda\nimport torch\ncuda.select_device(0)\ncuda.close()\ncuda.select_device(0)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:12.060936Z","iopub.execute_input":"2021-08-09T14:01:12.061558Z","iopub.status.idle":"2021-08-09T14:01:14.885413Z","shell.execute_reply.started":"2021-08-09T14:01:12.061513Z","shell.execute_reply":"2021-08-09T14:01:14.884573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## YOLOv5","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nimport torch\nimport warnings\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:14.886709Z","iopub.execute_input":"2021-08-09T14:01:14.887097Z","iopub.status.idle":"2021-08-09T14:01:15.134032Z","shell.execute_reply.started":"2021-08-09T14:01:14.887067Z","shell.execute_reply":"2021-08-09T14:01:15.133044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = meta[meta['split'] == 'test']\nif fast_sub:\n    test_df = fast_df.copy()\nelse:\n    test_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\n\ntest_df = df[study_len:].reset_index(drop=True)\nmeta.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:15.135441Z","iopub.execute_input":"2021-08-09T14:01:15.135803Z","iopub.status.idle":"2021-08-09T14:01:15.217657Z","shell.execute_reply.started":"2021-08-09T14:01:15.135736Z","shell.execute_reply":"2021-08-09T14:01:15.216847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta['image_id'] = meta['image_id'] + '_image'\nmeta.columns = ['id', 'dim0', 'dim1', 'split']\ntest_df = pd.merge(test_df, meta, on = 'id', how = 'left')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:15.218926Z","iopub.execute_input":"2021-08-09T14:01:15.219299Z","iopub.status.idle":"2021-08-09T14:01:15.305596Z","shell.execute_reply.started":"2021-08-09T14:01:15.219242Z","shell.execute_reply":"2021-08-09T14:01:15.304283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim = 512 #1024, 256, 'original'\ntest_dir = f'/kaggle/tmp/{split}/image'\n\n\n#weights_dir = '/kaggle/input/siimc19yolov5trainimage-object-detection/siimc19_yolo_best_512_b24_epoch35.pt'\nweights_dir = '/kaggle/input/siimc19-y5-only-opaque-image-object-detect-weights/siimc19_yolo_best_512_b24_epoch35_only_opaque.pt'\n\nshutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\nos.chdir('/kaggle/working/yolov5') # install dependencies\nprint('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:15.307469Z","iopub.execute_input":"2021-08-09T14:01:15.307965Z","iopub.status.idle":"2021-08-09T14:01:15.955532Z","shell.execute_reply.started":"2021-08-09T14:01:15.307925Z","shell.execute_reply":"2021-08-09T14:01:15.954635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights $weights_dir\\\n--img 512\\\n--conf 0.001\\\n--iou 0.5\\\n--source $test_dir\\\n--save-txt --save-conf --exist-ok --augment","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:01:15.957592Z","iopub.execute_input":"2021-08-09T14:01:15.957998Z","iopub.status.idle":"2021-08-09T14:04:15.398123Z","shell.execute_reply.started":"2021-08-09T14:01:15.957957Z","shell.execute_reply":"2021-08-09T14:04:15.396967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n\n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n\n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n\n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n\n    return bboxes","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:15.400056Z","iopub.execute_input":"2021-08-09T14:04:15.400506Z","iopub.status.idle":"2021-08-09T14:04:15.566287Z","shell.execute_reply.started":"2021-08-09T14:04:15.400461Z","shell.execute_reply":"2021-08-09T14:04:15.56527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = []\nPredictionStrings = []\npredbox_ens = []\nscore = []\nbox = []\n\nfor file_path in tqdm(glob('runs/detect/exp/labels/*.txt')):\n    \n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.id==image_id,['dim1', 'dim0']].values[0]\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    data = np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis=1)\n    \n    #import pdb;pdb.set_trace()\n    #bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 12).astype(str))\n    #for idx in range(len(bboxes)):\n    #    bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    \n    image_ids.append(image_id)\n    #PredictionStrings.append(' '.join(bboxes))\n    score.append(data[:,1])\n    box.append(data[:,2:])\n\npred_df_yolo = pd.DataFrame({'id':image_ids,'score':score,'label':0,'box':box})\npred_df_yolo.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:15.5695Z","iopub.execute_input":"2021-08-09T14:04:15.569784Z","iopub.status.idle":"2021-08-09T14:04:18.188384Z","shell.execute_reply.started":"2021-08-09T14:04:15.569758Z","shell.execute_reply":"2021-08-09T14:04:18.187588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MMDetection","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nimport torch","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.189654Z","iopub.execute_input":"2021-08-09T14:04:18.19004Z","iopub.status.idle":"2021-08-09T14:04:18.260066Z","shell.execute_reply.started":"2021-08-09T14:04:18.190003Z","shell.execute_reply":"2021-08-09T14:04:18.259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.261413Z","iopub.execute_input":"2021-08-09T14:04:18.261826Z","iopub.status.idle":"2021-08-09T14:04:18.335848Z","shell.execute_reply.started":"2021-08-09T14:04:18.261785Z","shell.execute_reply":"2021-08-09T14:04:18.334963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = meta[meta['split'] == 'test']\nif fast_sub:\n    test_df = fast_df.copy()\nelse:\n    test_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\n\ntest_df = df[study_len:].reset_index(drop=True) ","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.337302Z","iopub.execute_input":"2021-08-09T14:04:18.337715Z","iopub.status.idle":"2021-08-09T14:04:18.423242Z","shell.execute_reply.started":"2021-08-09T14:04:18.337672Z","shell.execute_reply":"2021-08-09T14:04:18.422402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.424731Z","iopub.execute_input":"2021-08-09T14:04:18.425111Z","iopub.status.idle":"2021-08-09T14:04:18.501197Z","shell.execute_reply.started":"2021-08-09T14:04:18.425075Z","shell.execute_reply":"2021-08-09T14:04:18.500211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.502829Z","iopub.execute_input":"2021-08-09T14:04:18.503318Z","iopub.status.idle":"2021-08-09T14:04:18.580342Z","shell.execute_reply.started":"2021-08-09T14:04:18.503278Z","shell.execute_reply":"2021-08-09T14:04:18.579074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#meta['image_id'] = meta['image_id'] + '_image'\nmeta.columns = ['id', 'dim0', 'dim1', 'split']\ntest_df = pd.merge(test_df, meta, on = 'id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.582018Z","iopub.execute_input":"2021-08-09T14:04:18.582472Z","iopub.status.idle":"2021-08-09T14:04:18.653725Z","shell.execute_reply.started":"2021-08-09T14:04:18.582428Z","shell.execute_reply":"2021-08-09T14:04:18.652803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/tmp","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:18.655378Z","iopub.execute_input":"2021-08-09T14:04:18.656063Z","iopub.status.idle":"2021-08-09T14:04:19.657617Z","shell.execute_reply.started":"2021-08-09T14:04:18.656018Z","shell.execute_reply":"2021-08-09T14:04:19.656542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dim = 512 #1024, 256, 'original'\ntest_dir = f'/kaggle/tmp/test/image'\n\n# #test_dir = f'/kaggle/input/siimc19512imgpngwithyololabels/test_images'\n# #weights_dir = '/kaggle/input/siim-cov19-yolov5-train/yolov5/runs/train/exp/weights/best.pt'\n\n# weights_dir = '/kaggle/input/siimc19yolov5trainimage-object-detection/siimc19_yolo_best_512_b24_epoch35.pt'\n\n# shutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\n# os.chdir('/kaggle/working/yolov5') # install dependencies\n\n# print('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:19.66116Z","iopub.execute_input":"2021-08-09T14:04:19.66151Z","iopub.status.idle":"2021-08-09T14:04:19.740461Z","shell.execute_reply.started":"2021-08-09T14:04:19.661477Z","shell.execute_reply":"2021-08-09T14:04:19.739552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\n\nimport torch\ndevice = torch.device(\"cuda\") if torch.cuda.is_available() else torch.device(\"cpu\")\nprint(device.type)\n\nimport torchvision\nprint(torch.__version__, torch.cuda.is_available())\n\n# Check mmcv installation\nfrom mmcv.ops import get_compiling_cuda_version, get_compiler_version\nprint(get_compiling_cuda_version())\nprint(get_compiler_version())","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:19.741766Z","iopub.execute_input":"2021-08-09T14:04:19.742101Z","iopub.status.idle":"2021-08-09T14:04:35.694263Z","shell.execute_reply.started":"2021-08-09T14:04:19.742067Z","shell.execute_reply":"2021-08-09T14:04:35.693182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install '/kaggle/input/weightedboxesfusion/' --no-deps\nfrom ensemble_boxes import weighted_boxes_fusion, nms","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:35.695924Z","iopub.execute_input":"2021-08-09T14:04:35.69621Z","iopub.status.idle":"2021-08-09T14:04:58.246569Z","shell.execute_reply.started":"2021-08-09T14:04:35.696182Z","shell.execute_reply":"2021-08-09T14:04:58.245598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imports\n%cd /kaggle/working/mmdetection\nimport mmdet\n\nfrom mmdet.apis import set_random_seed\nfrom mmdet.datasets import build_dataset\nfrom mmdet.models import build_detector\n# Check MMDetection installation\nfrom mmdet.apis import set_random_seed\n%cd /kaggle/working/\n\nimport mmcv\nfrom mmcv import Config\nfrom mmcv.runner import load_checkpoint\nfrom mmcv.parallel import MMDataParallel\nfrom mmdet.apis import inference_detector, init_detector, show_result_pyplot\nfrom mmdet.apis import single_gpu_test\nfrom mmdet.datasets import build_dataloader, build_dataset","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:04:58.250293Z","iopub.execute_input":"2021-08-09T14:04:58.250561Z","iopub.status.idle":"2021-08-09T14:05:08.114127Z","shell.execute_reply.started":"2021-08-09T14:04:58.250533Z","shell.execute_reply":"2021-08-09T14:05:08.113158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\nlabel2color = [[59, 238, 119]]\n\nviz_labels =  [\"Covid_Abnormality\"]\n\ndef plot_img(img, size=(18, 18), is_rgb=True, title=\"\", cmap=None):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n    \ndef plot_imgs(imgs, cols=2, size=10, is_rgb=True, title=\"\", cmap=None, img_size=None):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    return fig\n    \ndef draw_bbox(image, box, label, color):   \n    alpha = 0.1\n    alpha_font = 0.6\n    thickness = 8\n    font_size = 2.0\n    font_weight = 1\n    overlay_bbox = image.copy()\n    overlay_text = image.copy()\n    output = image.copy()\n\n    text_width, text_height = cv2.getTextSize(label.upper(), cv2.FONT_HERSHEY_SIMPLEX, font_size, font_weight)[0]\n    cv2.rectangle(overlay_bbox, (box[0], box[1]), (box[2], box[3]),\n                color, -1)\n    cv2.addWeighted(overlay_bbox, alpha, output, 1 - alpha, 0, output)\n    cv2.rectangle(overlay_text, (box[0], box[1]-18-text_height), (box[0]+text_width+8, box[1]),\n                (0, 0, 0), -1)\n    cv2.addWeighted(overlay_text, alpha_font, output, 1 - alpha_font, 0, output)\n    cv2.rectangle(output, (box[0], box[1]), (box[2], box[3]),\n                    color, thickness)\n    cv2.putText(output, label.upper(), (box[0], box[1]-12),\n            cv2.FONT_HERSHEY_SIMPLEX, font_size, (255, 255, 255), font_weight, cv2.LINE_AA)\n    return output\n\ndef draw_bbox_small(image, box, label, color):   \n    alpha = 0.1\n    alpha_text = 0.3\n    thickness = 1\n    font_size = 0.4\n    overlay_bbox = image.copy()\n    overlay_text = image.copy()\n    output = image.copy()\n\n    text_width, text_height = cv2.getTextSize(label.upper(), cv2.FONT_HERSHEY_SIMPLEX, font_size, thickness)[0]\n    cv2.rectangle(overlay_bbox, (box[0], box[1]), (box[2], box[3]),\n                color, -1)\n    cv2.addWeighted(overlay_bbox, alpha, output, 1 - alpha, 0, output)\n    cv2.rectangle(overlay_text, (box[0], box[1]-7-text_height), (box[0]+text_width+2, box[1]),\n                (0, 0, 0), -1)\n    cv2.addWeighted(overlay_text, alpha_text, output, 1 - alpha_text, 0, output)\n    cv2.rectangle(output, (box[0], box[1]), (box[2], box[3]),\n                    color, thickness)\n    cv2.putText(output, label.upper(), (box[0], box[1]-5),\n            cv2.FONT_HERSHEY_SIMPLEX, font_size, (255, 255, 255), thickness, cv2.LINE_AA)\n    return output","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:08.115503Z","iopub.execute_input":"2021-08-09T14:05:08.116054Z","iopub.status.idle":"2021-08-09T14:05:08.204589Z","shell.execute_reply.started":"2021-08-09T14:05:08.116015Z","shell.execute_reply":"2021-08-09T14:05:08.203735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_cfg_path = \"/kaggle/input/siim-mmdetection-cascadercnn-weight-bias/job4_cascade_rcnn_x101_32x4d_fpn_1x_fold0/job4_cascade_rcnn_x101_32x4d_fpn_1x_coco.py\"\ncfg = Config.fromfile(baseline_cfg_path)\n\ncfg.classes = (\"Covid_Abnormality\")\ncfg.data.test.img_prefix = ''\ncfg.data.test.classes = cfg.classes\n\n# cfg.model.roi_head.bbox_head.num_classes = 1\n# cfg.model.bbox_head.num_classes = 1\nfor head in cfg.model.roi_head.bbox_head:\n    head.num_classes = 1\n\n# Set seed thus the results are more reproducible\ncfg.seed = 211\nset_random_seed(211, deterministic=False)\ncfg.gpu_ids = [0]\n\ncfg.data.test.pipeline=[\n            dict(type='LoadImageFromFile'),\n            dict(\n                type='MultiScaleFlipAug',\n                img_scale=(1333, 800),\n                flip=False,\n                transforms=[\n                    dict(type='Resize', keep_ratio=True),\n                    dict(type='RandomFlip', direction='horizontal'),\n                    dict(\n                        type='Normalize',\n                        mean=[123.675, 116.28, 103.53],\n                        std=[58.395, 57.12, 57.375],\n                        to_rgb=True),\n                    dict(type='Pad', size_divisor=32),\n                    dict(type='DefaultFormatBundle'),\n                    dict(type='Collect', keys=['img'])\n                ])\n        ]\n\ncfg.test_pipeline = [\n            dict(type='LoadImageFromFile'),\n            dict(\n                type='MultiScaleFlipAug',\n                img_scale=(1333, 800),\n                flip=False,\n                transforms=[\n                    dict(type='Resize', keep_ratio=True),\n                    dict(type='RandomFlip', direction='horizontal'),\n                    dict(\n                        type='Normalize',\n                        mean=[123.675, 116.28, 103.53],\n                        std=[58.395, 57.12, 57.375],\n                        to_rgb=True),\n                    dict(type='Pad', size_divisor=32),\n                    dict(type='DefaultFormatBundle'),\n                    dict(type='Collect', keys=['img'])\n                ])\n        ]\n\n# cfg.data.samples_per_gpu = 4\n# cfg.data.workers_per_gpu = 4\n# cfg.model.test_cfg.nms.iou_threshold = 0.3\ncfg.model.test_cfg.rcnn.score_thr = 0.001\n\nWEIGHTS_FILE = '/kaggle/input/siim-mmdetection-cascadercnn-weight-bias/job4_cascade_rcnn_x101_32x4d_fpn_1x_fold0/epoch_10.pth'\noptions = dict(classes = (\"Covid_Abnormality\"))\nmodel = init_detector(cfg, WEIGHTS_FILE, device='cuda:0')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:08.207632Z","iopub.execute_input":"2021-08-09T14:05:08.208011Z","iopub.status.idle":"2021-08-09T14:05:20.346987Z","shell.execute_reply.started":"2021-08-09T14:05:08.20798Z","shell.execute_reply":"2021-08-09T14:05:20.346004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['image_path'] = '/kaggle/tmp/test/image/'+ test_df['id']+'.png'\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:20.348281Z","iopub.execute_input":"2021-08-09T14:05:20.34884Z","iopub.status.idle":"2021-08-09T14:05:20.442639Z","shell.execute_reply.started":"2021-08-09T14:05:20.348799Z","shell.execute_reply":"2021-08-09T14:05:20.441517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #from ensemble_boxes import weighted_boxes_fusion, nms\n# IMAGE_DIMS = (512, 512)\n# viz_images = []\n# results = []\n# score_threshold = cfg.model.test_cfg.rcnn.score_thr\n\n# def format_pred(boxes: np.ndarray, scores: np.ndarray, labels: np.ndarray) -> str:\n#     pred_strings = []\n#     label_str = ['opacity']\n#     for label, score, bbox in zip(labels, scores, boxes):\n#         xmin, ymin, xmax, ymax = bbox.astype(np.int64)\n#         pred_strings.append(f\"{label_str[int(label)]} {score:.16f} {xmin} {ymin} {xmax} {ymax}\")\n#     return \" \".join(pred_strings)\n\n# model.to(device)\n# model.eval()\n\n# viz_images = []\n\n# with torch.no_grad():\n#     for index, row in tqdm(test_df.iterrows(), total=test_df.shape[0]):\n#         original_H, original_W = (int(row.dim0), int(row.dim1))\n#         predictions = inference_detector(model, row.image_path)\n#         boxes, scores, labels = (list(), list(), list())\n\n#         for k, cls_result in enumerate(predictions):\n# #             print(\"cls_result\", cls_result)\n#             if cls_result.size != 0:\n#                 if len(labels)==0:\n#                     boxes = np.array(cls_result[:, :4])\n#                     scores = np.array(cls_result[:, 4])\n#                     labels = np.array([k]*len(cls_result[:, 4]))\n#                 else:    \n#                     boxes = np.concatenate((boxes, np.array(cls_result[:, :4])))\n#                     scores = np.concatenate((scores, np.array(cls_result[:, 4])))\n#                     labels = np.concatenate((labels, [k]*len(cls_result[:, 4])))\n                    \n#             if fast_sub:\n#                 img_viz = cv2.imread(row.image_path)\n#                 for box, label, score in zip(boxes, labels, scores):\n#                     color = label2color[int(label)]\n#                     img_viz = draw_bbox_small(img_viz, box.astype(np.int32), f'opacity_{score:.4f}', color)\n#                 viz_images.append(img_viz)\n\n#         indexes = np.where(scores > score_threshold)\n# #         print(indexes)\n#         boxes = boxes[indexes]\n#         scores = scores[indexes]\n#         labels = labels[indexes]\n\n#         if len(labels) != 0:\n#             h_ratio = original_H/IMAGE_DIMS[0]\n#             w_ratio = original_W/IMAGE_DIMS[1]\n#             boxes[:, [0, 2]] *= w_ratio\n#             boxes[:, [1, 3]] *= h_ratio\n\n#             result = {\n#                 \"id\": row.id,\n#                 \"PredictionString\": format_pred(\n#                     boxes, scores, labels\n#                 ),\n#             }\n\n#             results.append(result)\n\nIMAGE_DIMS = (512, 512)\nviz_images = []\nresults = []\nscore_threshold = cfg.model.test_cfg.rcnn.score_thr\n\ndef format_pred(boxes: np.ndarray, scores: np.ndarray, labels: np.ndarray) -> str:\n    pred_strings = []\n    label_str = ['opacity']\n    for label, score, bbox in zip(labels, scores, boxes):\n        xmin, ymin, xmax, ymax = bbox.astype(np.int64)\n        pred_strings.append(f\"{label_str[int(label)]} {score:.16f} {xmin} {ymin} {xmax} {ymax}\")\n    return \" \".join(pred_strings)\n\nmodel.to(device)\nmodel.eval()\n\nviz_images = []\n\nwith torch.no_grad():\n    for index, row in tqdm(test_df.iterrows(), total=test_df.shape[0]):\n        original_H, original_W = (int(row.dim0), int(row.dim1))\n        predictions = inference_detector(model, row.image_path)\n        boxes, scores, labels = (list(), list(), list())\n\n        for k, cls_result in enumerate(predictions):\n#             print(\"cls_result\", cls_result)\n            if cls_result.size != 0:\n                if len(labels)==0:\n                    boxes = np.array(cls_result[:, :4])\n                    scores = np.array(cls_result[:, 4])\n                    labels = np.array([k]*len(cls_result[:, 4]))\n                else:    \n                    boxes = np.concatenate((boxes, np.array(cls_result[:, :4])))\n                    scores = np.concatenate((scores, np.array(cls_result[:, 4])))\n                    labels = np.concatenate((labels, [k]*len(cls_result[:, 4])))\n                    \n#             if fast_sub:\n#                 img_viz = cv2.imread(row.image_path)\n#                 for box, label, score in zip(boxes, labels, scores):\n#                     color = label2color[int(label)]\n#                     img_viz = draw_bbox_small(img_viz, box.astype(np.int32), f'opacity_{score:.4f}', color)\n#                 viz_images.append(img_viz)\n\n        indexes = np.where(scores > score_threshold)\n#         print(indexes)\n        boxes = boxes[indexes]\n        scores = scores[indexes]\n        labels = labels[indexes]\n\n        if len(labels) != 0:\n            h_ratio = original_H/IMAGE_DIMS[0]\n            w_ratio = original_W/IMAGE_DIMS[1]\n            boxes[:, [0, 2]] *= w_ratio\n            boxes[:, [1, 3]] *= h_ratio\n            \n            #-----------------------------------------------#\n            ##pred_df_yolo = pd.DataFrame({'id':image_ids,'score':score,'label':0,'box':box})\n            #import pdb\n            index = pred_df_yolo[pred_df_yolo.id==row.id].index\n            y_row = pred_df_yolo.iloc[index]\n            \n            m_norm_boxes = np.linalg.norm(boxes)\n            boxes_m = (boxes/m_norm_boxes).tolist()\n            scores_m = scores.tolist()\n            labels_m = labels.tolist()\n            \n            boxes_y = []\n            scores_y = []\n            labels_y = []\n            #pdb.set_trace()\n            if len(y_row.score) != 0:\n                boxes_y = (y_row.box.values[0]/m_norm_boxes).tolist()            \n                scores_y = y_row.score.values[0].tolist()\n                labels_y = np.zeros_like(scores_y, dtype=np.int).tolist()\n            \n            ens_boxes = [boxes_m, boxes_y]\n            ens_score = [scores_m, scores_y]\n            ens_label = [labels_m, labels_y]\n            weight_detyolo = [2,1]\n            \n            boxes, scores, labels  = weighted_boxes_fusion(ens_boxes, ens_score, ens_label,\n                                                           weights=weight_detyolo, iou_thr=0.6,\n                                                           conf_type='avg',skip_box_thr=0.001)\n            \n            #boxes, scores, labels = nms([boxes], [scores], [labels], weights=None, iou_thr=0.5)\n            \n            #pdb.set_trace()\n            boxes = boxes * m_norm_boxes\n            #-----------------------------------------------#\n            \n            result = {\n                \"id\": row.id,\n                \"PredictionString\": format_pred(\n                    boxes, scores, labels\n                ),\n            }\n\n            results.append(result)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:05:20.444277Z","iopub.execute_input":"2021-08-09T14:05:20.444718Z","iopub.status.idle":"2021-08-09T14:08:39.288619Z","shell.execute_reply.started":"2021-08-09T14:05:20.444681Z","shell.execute_reply":"2021-08-09T14:08:39.287847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del model\ngc.collect()\n\npred_df = pd.DataFrame(results, columns=['id', 'PredictionString'])\n\nif fast_sub:\n    display(pred_df.sample(2))\n    # Plot sample images\n    #plot_imgs(viz_images, cmap=None)\n    #plt.savefig('viz_fig_siim.png', bbox_inches='tight')\n    #plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:39.294491Z","iopub.execute_input":"2021-08-09T14:08:39.29847Z","iopub.status.idle":"2021-08-09T14:08:39.779665Z","shell.execute_reply.started":"2021-08-09T14:08:39.298429Z","shell.execute_reply":"2021-08-09T14:08:39.77885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:39.783641Z","iopub.execute_input":"2021-08-09T14:08:39.785676Z","iopub.status.idle":"2021-08-09T14:08:39.911023Z","shell.execute_reply.started":"2021-08-09T14:08:39.785637Z","shell.execute_reply":"2021-08-09T14:08:39.910278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(['PredictionString','image_path'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:39.914684Z","iopub.execute_input":"2021-08-09T14:08:39.916709Z","iopub.status.idle":"2021-08-09T14:08:40.006495Z","shell.execute_reply.started":"2021-08-09T14:08:39.916666Z","shell.execute_reply":"2021-08-09T14:08:40.005457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:40.008003Z","iopub.execute_input":"2021-08-09T14:08:40.008595Z","iopub.status.idle":"2021-08-09T14:08:40.092182Z","shell.execute_reply.started":"2021-08-09T14:08:40.008555Z","shell.execute_reply":"2021-08-09T14:08:40.091077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.merge(test_df, pred_df, on = 'id', how = 'left').fillna(\"none 1 0 0 1 1\")\nsub_df = sub_df[['id', 'PredictionString']]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:40.094166Z","iopub.execute_input":"2021-08-09T14:08:40.094651Z","iopub.status.idle":"2021-08-09T14:08:40.179461Z","shell.execute_reply.started":"2021-08-09T14:08:40.094611Z","shell.execute_reply":"2021-08-09T14:08:40.178532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['none'] = df_2class['none']","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:40.181104Z","iopub.execute_input":"2021-08-09T14:08:40.181482Z","iopub.status.idle":"2021-08-09T14:08:40.254996Z","shell.execute_reply.started":"2021-08-09T14:08:40.181445Z","shell.execute_reply":"2021-08-09T14:08:40.254016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] == \"none 1 0 0 1 1\":\n        continue\n    sub_df_split = sub_df.loc[i,'PredictionString'].split()\n    sub_df_list = []\n    for j in range(int(len(sub_df_split) / 6)):\n        sub_df_list.append('opacity')\n        #sub_df_list.append(sub_df_split[6 * j + 1])\n        x = float(sub_df_split[6 * j + 1])*(1.0-sub_df.loc[i,'none'])\n        #sub_df_list.append(sub_df_split[6 * j + 1]*(1.0-sub_df.loc[i,'none']))\n        sub_df_list.append(str(x))\n        sub_df_list.append(sub_df_split[6 * j + 2])\n        sub_df_list.append(sub_df_split[6 * j + 3])\n        sub_df_list.append(sub_df_split[6 * j + 4])\n        sub_df_list.append(sub_df_split[6 * j + 5])\n    sub_df.loc[i,'PredictionString'] = ' '.join(sub_df_list)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:40.256579Z","iopub.execute_input":"2021-08-09T14:08:40.257283Z","iopub.status.idle":"2021-08-09T14:08:42.02766Z","shell.execute_reply.started":"2021-08-09T14:08:40.257231Z","shell.execute_reply":"2021-08-09T14:08:42.026783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] != 'none 1 0 0 1 1':\n        sub_df.loc[i,'PredictionString'] = sub_df.loc[i,'PredictionString'] + ' none ' + str(sub_df.loc[i,'none']) + ' 0 0 1 1'","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:42.029083Z","iopub.execute_input":"2021-08-09T14:08:42.029431Z","iopub.status.idle":"2021-08-09T14:08:42.665119Z","shell.execute_reply.started":"2021-08-09T14:08:42.029395Z","shell.execute_reply":"2021-08-09T14:08:42.664235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = sub_df[['id', 'PredictionString']]   \nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:42.671149Z","iopub.execute_input":"2021-08-09T14:08:42.671414Z","iopub.status.idle":"2021-08-09T14:08:42.752381Z","shell.execute_reply.started":"2021-08-09T14:08:42.671388Z","shell.execute_reply":"2021-08-09T14:08:42.751423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_study = df_study[:study_len]\n# df_study = df_study.append(sub_df).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:42.754854Z","iopub.execute_input":"2021-08-09T14:08:42.755115Z","iopub.status.idle":"2021-08-09T14:08:42.828143Z","shell.execute_reply.started":"2021-08-09T14:08:42.755091Z","shell.execute_reply":"2021-08-09T14:08:42.827068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df = study_df[['id', 'PredictionString']]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:42.829851Z","iopub.execute_input":"2021-08-09T14:08:42.830242Z","iopub.status.idle":"2021-08-09T14:08:42.902107Z","shell.execute_reply.started":"2021-08-09T14:08:42.830205Z","shell.execute_reply":"2021-08-09T14:08:42.901213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study = study_df.append(sub_df).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:42.903433Z","iopub.execute_input":"2021-08-09T14:08:42.903819Z","iopub.status.idle":"2021-08-09T14:08:42.974539Z","shell.execute_reply.started":"2021-08-09T14:08:42.903777Z","shell.execute_reply":"2021-08-09T14:08:42.973717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:42.975674Z","iopub.execute_input":"2021-08-09T14:08:42.976148Z","iopub.status.idle":"2021-08-09T14:08:43.052046Z","shell.execute_reply.started":"2021-08-09T14:08:42.976117Z","shell.execute_reply":"2021-08-09T14:08:43.050897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study.to_csv('/kaggle/working/submission.csv',index = False)  \nshutil.rmtree('/kaggle/working/mmdetection')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:43.05385Z","iopub.execute_input":"2021-08-09T14:08:43.054314Z","iopub.status.idle":"2021-08-09T14:08:43.262651Z","shell.execute_reply.started":"2021-08-09T14:08:43.054276Z","shell.execute_reply":"2021-08-09T14:08:43.261886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.rmtree('/kaggle/working/yolov5')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T14:08:43.263996Z","iopub.execute_input":"2021-08-09T14:08:43.264357Z","iopub.status.idle":"2021-08-09T14:08:43.440049Z","shell.execute_reply.started":"2021-08-09T14:08:43.264321Z","shell.execute_reply":"2021-08-09T14:08:43.439193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}