{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<span style=\"color: #027fc1; font-family: Segoe UI; font-size: 1.9em; font-weight: 300;\">Overview</span>\n\nThis notebook is forked from https://www.kaggle.com/h053473666/siim-cov19-efnb7-yolov5-infer.\n\nI replaced YOLOv5 inference with **YOLOv4**.\n\nMain challenge was to install YOLOv4 on Kaggle kernel. \n\nI trained YOLOv4 for 10 000 iterations. \n\n<span style=\"color: #000508; font-family: Segoe UI; font-size: 1.1em;\">Results on best weights:</span><br>\n✅ maP is **79.98%**<br>\n✅ IoU is **58.42%**\n\nYou may change hyperparameters in YOLOv4 inference section to get better scores.","metadata":{}},{"cell_type":"markdown","source":"## Install libraries to work with .dcm","metadata":{}},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-18T10:11:53.176449Z","iopub.execute_input":"2021-07-18T10:11:53.177122Z","iopub.status.idle":"2021-07-18T10:13:18.357006Z","shell.execute_reply.started":"2021-07-18T10:11:53.176951Z","shell.execute_reply":"2021-07-18T10:13:18.355652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-18T10:13:18.360584Z","iopub.execute_input":"2021-07-18T10:13:18.361054Z","iopub.status.idle":"2021-07-18T10:13:18.374571Z","shell.execute_reply.started":"2021-07-18T10:13:18.360981Z","shell.execute_reply":"2021-07-18T10:13:18.373474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fast_sub = True\ndf = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nif df.shape[0] == 2477:\n    fast_sub = True\n    fast_df = pd.DataFrame(([['00086460a852_study', 'negative 1 0 0 1 1'], \n                         ['000c9c05fd14_study', 'negative 1 0 0 1 1'], \n                         ['65761e66de9f_image', 'none 1 0 0 1 1'], \n                         ['51759b5579bc_image', 'none 1 0 0 1 1']]), \n                       columns=['id', 'PredictionString'])\nelse:\n    fast_sub = False","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:18.376961Z","iopub.execute_input":"2021-07-18T10:13:18.377453Z","iopub.status.idle":"2021-07-18T10:13:18.406616Z","shell.execute_reply.started":"2021-07-18T10:13:18.37741Z","shell.execute_reply":"2021-07-18T10:13:18.405542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fast_sub = False","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:18.421267Z","iopub.execute_input":"2021-07-18T10:13:18.422164Z","iopub.status.idle":"2021-07-18T10:13:18.427623Z","shell.execute_reply.started":"2021-07-18T10:13:18.422118Z","shell.execute_reply":"2021-07-18T10:13:18.42627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# .dcm to .png","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:18.429659Z","iopub.execute_input":"2021-07-18T10:13:18.430777Z","iopub.status.idle":"2021-07-18T10:13:18.748969Z","shell.execute_reply.started":"2021-07-18T10:13:18.430729Z","shell.execute_reply":"2021-07-18T10:13:18.747841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:18.752788Z","iopub.execute_input":"2021-07-18T10:13:18.753167Z","iopub.status.idle":"2021-07-18T10:13:18.760774Z","shell.execute_reply.started":"2021-07-18T10:13:18.753129Z","shell.execute_reply":"2021-07-18T10:13:18.758109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create images for classification (EfficientNet)","metadata":{}},{"cell_type":"code","source":"split = 'test'\nsave_dir = f'/kaggle/tmp/{split}/'\n\nos.makedirs(save_dir, exist_ok=True)\n\nsave_dir = f'/kaggle/tmp/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=600)  \n    study = '00086460a852' + '_study.png'\n    im.save(os.path.join(save_dir, study))\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=600)  \n    study = '000c9c05fd14' + '_study.png'\n    im.save(os.path.join(save_dir, study))\nelse:   \n    for dirname, _, filenames in tqdm(os.walk(f'/kaggle/input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=600)  \n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:18.765968Z","iopub.execute_input":"2021-07-18T10:13:18.76634Z","iopub.status.idle":"2021-07-18T10:13:20.481312Z","shell.execute_reply.started":"2021-07-18T10:13:18.766309Z","shell.execute_reply":"2021-07-18T10:13:20.480213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\nsave_dir = f'/kaggle/tmp/{split}/image/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'/kaggle/input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=512)  \n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)\nmeta = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:20.483824Z","iopub.execute_input":"2021-07-18T10:13:20.484156Z","iopub.status.idle":"2021-07-18T10:13:20.933853Z","shell.execute_reply.started":"2021-07-18T10:13:20.484124Z","shell.execute_reply":"2021-07-18T10:13:20.932593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create images for detection (YOLOv4)\n\nI don't resize images because I trained on original images.","metadata":{}},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\nsave_dir = f'/kaggle/tmp/{split}/image_yolo4/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('/kaggle/input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'/kaggle/input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = Image.fromarray(xray)\n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:20.935495Z","iopub.execute_input":"2021-07-18T10:13:20.935934Z","iopub.status.idle":"2021-07-18T10:13:21.383134Z","shell.execute_reply.started":"2021-07-18T10:13:20.935886Z","shell.execute_reply":"2021-07-18T10:13:21.381804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study predict","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nif fast_sub:\n    df = fast_df.copy()\nelse:\n    df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\n\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-18T10:13:21.384798Z","iopub.execute_input":"2021-07-18T10:13:21.385287Z","iopub.status.idle":"2021-07-18T10:13:21.402588Z","shell.execute_reply.started":"2021-07-18T10:13:21.385242Z","shell.execute_reply":"2021-07-18T10:13:21.401186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/kerasapplications -q\n!pip install /kaggle/input/efficientnet-keras-source-code/ -q --no-deps\n\nimport os\n\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport random\n\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n\n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 355.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n\n    def decode_with_labels(path, label):\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        \n        random_number = random.randint(0,2)\n        if random_number == 1:\n            img = tf.image.adjust_brightness(img, 1.2)\n        if random_number == 2:\n            img = tf.image.adjust_brightness(img, 0.8)\n        return img\n\n    def augment_with_labels(img, label):\n        return augment(img), label\n\n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n\n    return dset\n\n#COMPETITION_NAME = \"siim-cov19-test-img512-study-600\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\n\nIMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512)\n\n#load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\nif fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[:study_len]\ntest_paths = f'/kaggle/tmp/{split}/study/' + sub_df['id'] +'.png'\n\nsub_df['negative'] = 0\nsub_df['typical'] = 0\nsub_df['indeterminate'] = 0\nsub_df['atypical'] = 0\n\n\nlabel_cols = sub_df.columns[2:]\n\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)\n\nwith strategy.scope():\n    \n    models = []\n    \n    models0 = tf.keras.models.load_model(\n        '../input/siim-covid19-efnb7-train-study/model0.h5'\n    )\n    models1 = tf.keras.models.load_model(\n        '../input/siim-covid19-efnb7-train-study/model1.h5'\n    )\n    models2 = tf.keras.models.load_model(\n        '../input/siim-covid19-efnb7-train-study/model2.h5'\n    )\n    models3 = tf.keras.models.load_model(\n        '../input/siim-covid19-efnb7-train-study/model3.h5'\n    )\n    models4 = tf.keras.models.load_model(\n        '../input/siim-covid19-efnb7-train-study/model4.h5'\n    )\n    \n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    models.append(models4)\n    \nweights = {\n    0: 1,\n    1: 1,\n    2: 1,\n    3: 1,\n    4: 3\n}\n\nweights_sum = sum(weights.values())\nweights = {k: v/weights_sum for k, v in weights.items()}\n\npredictions = [model.predict(dtest, verbose=1) for model in models]\nfor i, pred in enumerate(predictions):\n    predictions[i] = weights[i] * pred\n    \nsub_df[label_cols] = sum(predictions)    \n\n#sub_df[label_cols] = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:13:21.404425Z","iopub.execute_input":"2021-07-18T10:13:21.405179Z","iopub.status.idle":"2021-07-18T10:16:55.294321Z","shell.execute_reply.started":"2021-07-18T10:13:21.405132Z","shell.execute_reply":"2021-07-18T10:16:55.293145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.columns = ['id', 'PredictionString1', 'negative', 'typical', 'indeterminate', 'atypical']\ndf = pd.merge(df, sub_df, on = 'id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:16:55.296192Z","iopub.execute_input":"2021-07-18T10:16:55.2969Z","iopub.status.idle":"2021-07-18T10:16:55.311583Z","shell.execute_reply.started":"2021-07-18T10:16:55.296839Z","shell.execute_reply":"2021-07-18T10:16:55.310158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"for i in range(study_len):\n    negative = df.loc[i,'negative']\n    typical = df.loc[i,'typical']\n    indeterminate = df.loc[i,'indeterminate']\n    atypical = df.loc[i,'atypical']\n    df.loc[i, 'PredictionString'] = f'negative {negative} 0 0 1 1 typical {typical} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1'","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:16:55.346326Z","iopub.execute_input":"2021-07-18T10:16:55.346781Z","iopub.status.idle":"2021-07-18T10:16:55.356654Z","shell.execute_reply.started":"2021-07-18T10:16:55.346734Z","shell.execute_reply":"2021-07-18T10:16:55.354854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study = df[['id', 'PredictionString']]\n\n# df.to_csv('submission.csv',index=False)\n# df","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:16:55.358912Z","iopub.execute_input":"2021-07-18T10:16:55.35948Z","iopub.status.idle":"2021-07-18T10:16:55.369558Z","shell.execute_reply.started":"2021-07-18T10:16:55.359432Z","shell.execute_reply":"2021-07-18T10:16:55.36811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 class","metadata":{}},{"cell_type":"code","source":"if fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[study_len:]\ntest_paths = f'/kaggle/tmp/{split}/image/' + sub_df['id'] +'.png'\nsub_df['none'] = 0\n\nlabel_cols = sub_df.columns[2]\n\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[8], IMSIZE[8]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)\n\nwith strategy.scope():\n    \n    models = []\n    \n    models0 = tf.keras.models.load_model(\n        '/kaggle/input/siim-covid19-efnb7-train-fold0-5-2class/model0.h5'\n    )\n    models1 = tf.keras.models.load_model(\n        '/kaggle/input/siim-covid19-efnb7-train-fold0-5-2class/model1.h5'\n    )\n    models2 = tf.keras.models.load_model(\n        '/kaggle/input/siim-covid19-efnb7-train-fold0-5-2class/model2.h5'\n    )\n    models3 = tf.keras.models.load_model(\n        '/kaggle/input/siim-covid19-efnb7-train-fold0-5-2class/model3.h5'\n    )\n    models4 = tf.keras.models.load_model(\n        '/kaggle/input/siim-covid19-efnb7-train-fold0-5-2class/model4.h5'\n    )\n    \n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    models.append(models4)\n\nweights = {\n    0: 1,\n    1: 1,\n    2: 1,\n    3: 1,\n    4: 3\n}\n\nweights_sum = sum(weights.values())\nweights = {k: v/weights_sum for k, v in weights.items()}\n\npredictions = [model.predict(dtest, verbose=1) for model in models]\nfor i, pred in enumerate(predictions):\n    predictions[i] = weights[i] * pred\n    \nsub_df[label_cols] = sum(predictions)\n#sub_df[label_cols] = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)\ndf_2class = sub_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:16:55.371417Z","iopub.execute_input":"2021-07-18T10:16:55.372066Z","iopub.status.idle":"2021-07-18T10:19:22.925763Z","shell.execute_reply.started":"2021-07-18T10:16:55.371997Z","shell.execute_reply":"2021-07-18T10:19:22.924608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del models\ndel models0, models1, models2, models3, models4","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:22.927434Z","iopub.execute_input":"2021-07-18T10:19:22.927925Z","iopub.status.idle":"2021-07-18T10:19:22.936053Z","shell.execute_reply.started":"2021-07-18T10:19:22.927879Z","shell.execute_reply":"2021-07-18T10:19:22.934539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numba import cuda\nimport torch\ncuda.select_device(0)\ncuda.close()\ncuda.select_device(0)","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:22.937672Z","iopub.execute_input":"2021-07-18T10:19:22.938445Z","iopub.status.idle":"2021-07-18T10:19:25.794177Z","shell.execute_reply.started":"2021-07-18T10:19:22.938385Z","shell.execute_reply":"2021-07-18T10:19:25.793076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# YOLOv4 installation","metadata":{}},{"cell_type":"code","source":"import shutil\nimport os","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:25.79597Z","iopub.execute_input":"2021-07-18T10:19:25.796455Z","iopub.status.idle":"2021-07-18T10:19:25.801518Z","shell.execute_reply.started":"2021-07-18T10:19:25.796407Z","shell.execute_reply":"2021-07-18T10:19:25.800139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy darknet repository (https://github.com/AlexeyAB/darknet)\n\n# I changed Makefile as follows:\n# OPENCV=1        enable OpenCV\n# GPU=1           enable GPU\n# CUDNN=1         enable cuDNN\n# LIBSO=1         need to create libdarknet.so\n# CUDNN_HALF=1    enable mixed calculations\n\nshutil.copytree('/kaggle/input/darknetrepo3/darknet', '/kaggle/working/darknet')","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:25.803375Z","iopub.execute_input":"2021-07-18T10:19:25.803997Z","iopub.status.idle":"2021-07-18T10:19:36.722086Z","shell.execute_reply.started":"2021-07-18T10:19:25.803952Z","shell.execute_reply":"2021-07-18T10:19:36.720882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# verify CUDA\n!/usr/local/cuda/bin/nvcc --version","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:36.725635Z","iopub.execute_input":"2021-07-18T10:19:36.725949Z","iopub.status.idle":"2021-07-18T10:19:37.574466Z","shell.execute_reply.started":"2021-07-18T10:19:36.725916Z","shell.execute_reply":"2021-07-18T10:19:37.573164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I don't know why but I get error during make installation with kaggle kernel libcuda.so.\n# Therefore I replaced it with my own libcuda.so.\n\n%cd /usr/lib/gcc/x86_64-linux-gnu/7/../../../x86_64-linux-gnu/\n!rm libcuda.so\nshutil.copytree('/kaggle/input/libcuda', '/kaggle/working/libcuda')\n!cp '/kaggle/working/libcuda/libcuda.so' .","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:37.581595Z","iopub.execute_input":"2021-07-18T10:19:37.581915Z","iopub.status.idle":"2021-07-18T10:19:40.42366Z","shell.execute_reply.started":"2021-07-18T10:19:37.581882Z","shell.execute_reply":"2021-07-18T10:19:40.418432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# install darknet\n%cd /kaggle/working/darknet\n!make","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:19:40.433561Z","iopub.execute_input":"2021-07-18T10:19:40.434237Z","iopub.status.idle":"2021-07-18T10:21:38.014887Z","shell.execute_reply.started":"2021-07-18T10:19:40.434178Z","shell.execute_reply":"2021-07-18T10:21:38.013515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that libdarknet.so has been created\n!ls","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:38.017109Z","iopub.execute_input":"2021-07-18T10:21:38.017568Z","iopub.status.idle":"2021-07-18T10:21:38.859632Z","shell.execute_reply.started":"2021-07-18T10:21:38.017518Z","shell.execute_reply":"2021-07-18T10:21:38.858286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## YOLOv4 Detection","metadata":{}},{"cell_type":"code","source":"# copy data for YOLOv4 inferrence\n\n# obj.names - list of classes\n# obj.data - classes count and path to obj.names\n# yolov4-obj-mycustom.cfg - config\n# yolov4-obj-mycustom_last.weights - weights\n# my_darknet.py - script for inference. I added path to libdarknet.so and supportive procedure for inference\n\nshutil.copytree('/kaggle/input/yolo4data', '/kaggle/working/yolo4data')","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:38.861967Z","iopub.execute_input":"2021-07-18T10:21:38.862488Z","iopub.status.idle":"2021-07-18T10:21:43.107615Z","shell.execute_reply.started":"2021-07-18T10:21:38.862436Z","shell.execute_reply":"2021-07-18T10:21:43.106516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nfrom skimage import exposure\nfrom tqdm import tqdm\nequalize_hist = True\n\nfrom yolo4data.my_darknet2 import load_network, detect_image\n\nconfig_path = \"/kaggle/working/yolo4data/yolov4-obj-mycustom_672.cfg\"\nweight_path = \"/kaggle/working/yolo4data/yolov4-obj-mycustom_last.weights\"\nmeta_path = \"/kaggle/working/yolo4data/obj.data\"\n\n# create model\nnet_main, class_names, colors = load_network(config_path, meta_path, weight_path)","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:43.110003Z","iopub.execute_input":"2021-07-18T10:21:43.11036Z","iopub.status.idle":"2021-07-18T10:21:54.357058Z","shell.execute_reply.started":"2021-07-18T10:21:43.110327Z","shell.execute_reply":"2021-07-18T10:21:54.355719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np, pandas as pd\n\nmeta = meta[meta['split'] == 'test']\nif fast_sub:\n    test_df = fast_df.copy()\nelse:\n    test_df = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv')\ntest_df = df[study_len:].reset_index(drop=True) \nmeta['image_id'] = meta['image_id'] + '_image'\nmeta.columns = ['id', 'dim0', 'dim1', 'split']\ntest_df = pd.merge(test_df, meta, on = 'id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.358838Z","iopub.execute_input":"2021-07-18T10:21:54.359341Z","iopub.status.idle":"2021-07-18T10:21:54.37715Z","shell.execute_reply.started":"2021-07-18T10:21:54.359292Z","shell.execute_reply":"2021-07-18T10:21:54.375528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get images list\nsplit = 'test'\ntest_dir = f'/kaggle/tmp/{split}/image'\n\ndef get_all_files_in_folder(folder, types):\n    files_grabbed = []\n    for t in types:\n        files_grabbed.extend(folder.rglob(t))\n    files_grabbed = sorted(files_grabbed, key=lambda x: x)\n    return files_grabbed\n\nfrom pathlib import Path\nall_images = get_all_files_in_folder(Path(f'/kaggle/tmp/{split}/image_yolo4/'), ['*.png'])\nprint(len(all_images))","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.379155Z","iopub.execute_input":"2021-07-18T10:21:54.37972Z","iopub.status.idle":"2021-07-18T10:21:54.396999Z","shell.execute_reply.started":"2021-07-18T10:21:54.379672Z","shell.execute_reply":"2021-07-18T10:21:54.395787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n\n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n\n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n\n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n\n    return bboxes","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.398618Z","iopub.execute_input":"2021-07-18T10:21:54.399401Z","iopub.status.idle":"2021-07-18T10:21:54.409013Z","shell.execute_reply.started":"2021-07-18T10:21:54.399354Z","shell.execute_reply":"2021-07-18T10:21:54.407498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inference\n\nthreshold = .001\nhier_thresh = .5\nnms_coeff = .5\n\n\nimage_ids = []\nPredictionStrings = []\nfor image in tqdm(all_images):\n\n    imgArray = cv2.imread(str(image))\n    imgArray = cv2.cvtColor(imgArray, cv2.COLOR_BGR2RGB)\n    h, w = imgArray.shape[:2]\n    \n    \n    if equalize_hist:\n        xray_eq = exposure.equalize_hist(imgArray)\n        imgArray = ((xray_eq - xray_eq.min()) * (1 / (xray_eq.max() - xray_eq.min()) * 255)).astype('uint8')\n\n    detections = detect_image(net_main, class_names, imgArray, thresh=threshold, hier_thresh=hier_thresh, nms=nms_coeff)  # Class detection\n    \n    dect_results = []\n    for detection in detections:\n        if float(detection[1]) / 100 > threshold:\n            current_class = float(detection[0])\n            current_thresh = float(detection[1]) / 100\n            current_coords = [float(x) for x in detection[2]]\n            x_center_norm = current_coords[0] / w\n            y_center_norm = current_coords[1] / h\n            wn = current_coords[2] / w\n            hn = current_coords[3] / h\n            \n            d = [current_class, current_thresh, x_center_norm, y_center_norm, wn, hn]\n            dect_results.append(d)\n    \n    \n    if len(dect_results):\n        data = np.array(dect_results)\n        bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 12).astype(str))\n        for idx in range(len(bboxes)):\n            bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n        PredictionStrings.append(' '.join(bboxes))\n        image_ids.append(image.stem)","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.410993Z","iopub.execute_input":"2021-07-18T10:21:54.41171Z","iopub.status.idle":"2021-07-18T10:21:54.738822Z","shell.execute_reply.started":"2021-07-18T10:21:54.411663Z","shell.execute_reply":"2021-07-18T10:21:54.736948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\npred_df = pd.DataFrame({'id':image_ids, 'PredictionString':PredictionStrings})","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.740556Z","iopub.execute_input":"2021-07-18T10:21:54.741052Z","iopub.status.idle":"2021-07-18T10:21:54.753474Z","shell.execute_reply.started":"2021-07-18T10:21:54.740983Z","shell.execute_reply":"2021-07-18T10:21:54.746473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.782675Z","iopub.execute_input":"2021-07-18T10:21:54.783361Z","iopub.status.idle":"2021-07-18T10:21:54.79347Z","shell.execute_reply.started":"2021-07-18T10:21:54.783312Z","shell.execute_reply":"2021-07-18T10:21:54.791998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(['PredictionString'], axis=1)\nsub_df = pd.merge(test_df, pred_df, on = 'id', how = 'left').fillna(\"none 1 0 0 1 1\")\nsub_df = sub_df[['id', 'PredictionString']]\nfor i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] == \"none 1 0 0 1 1\":\n        continue\n    sub_df_split = sub_df.loc[i,'PredictionString'].split()\n    sub_df_list = []\n    for j in range(int(len(sub_df_split) / 6)):\n        sub_df_list.append('opacity')\n        sub_df_list.append(sub_df_split[6 * j + 1])\n        sub_df_list.append(sub_df_split[6 * j + 2])\n        sub_df_list.append(sub_df_split[6 * j + 3])\n        sub_df_list.append(sub_df_split[6 * j + 4])\n        sub_df_list.append(sub_df_split[6 * j + 5])\n    sub_df.loc[i,'PredictionString'] = ' '.join(sub_df_list)\nsub_df['none'] = df_2class['none'] \nfor i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] != 'none 1 0 0 1 1':\n        sub_df.loc[i,'PredictionString'] = sub_df.loc[i,'PredictionString'] + ' none ' + str(sub_df.loc[i,'none']) + ' 0 0 1 1'\nsub_df = sub_df[['id', 'PredictionString']]   \ndf_study = df_study[:study_len]\ndf_study = df_study.append(sub_df).reset_index(drop=True)\ndf_study.to_csv('/kaggle/working/submission.csv',index = False)  \n# shutil.rmtree('/kaggle/working/yolov5')","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:54.795609Z","iopub.execute_input":"2021-07-18T10:21:54.796205Z","iopub.status.idle":"2021-07-18T10:21:55.005327Z","shell.execute_reply.started":"2021-07-18T10:21:54.796157Z","shell.execute_reply":"2021-07-18T10:21:55.00423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.rmtree('/kaggle/working/darknet')\nshutil.rmtree('/kaggle/working/libcuda')\nshutil.rmtree('/kaggle/working/yolo4data')","metadata":{"execution":{"iopub.status.busy":"2021-07-18T10:21:55.006976Z","iopub.execute_input":"2021-07-18T10:21:55.007481Z","iopub.status.idle":"2021-07-18T10:21:55.141263Z","shell.execute_reply.started":"2021-07-18T10:21:55.007435Z","shell.execute_reply":"2021-07-18T10:21:55.140071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style='text-align: center;'><span style=\"color: #000508; font-family: Segoe UI; font-size: 2.0em; font-weight: 300;\">HAVE A GREAT DAY!</span></p>\n\n<p style='text-align: center;'><span style=\"color: #000508; font-family: Segoe UI; font-size: 1.3em; font-weight: 300;\">Let me know if you have any suggestions!</span></p>","metadata":{}}]}