{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"thanks to https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px  \nthanks to https://www.kaggle.com/awsaf49/vinbigdata-cxr-ad-yolov5-14-class-infer  \ntrain_study: https://www.kaggle.com/h053473666/siim-covid19-efnb7-train-study  \ntrain_image: https://www.kaggle.com/h053473666/siim-cov19-yolov5-train  \ntrain_2class: https://www.kaggle.com/h053473666/siim-covid19-efnb7-train-fold0-5-2class  \n  \nversion1:Original hyperparameters (yolov5)  \nversion4:New hyperparameters (yolov5)\n","metadata":{}},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-01T19:37:45.781453Z","iopub.execute_input":"2021-08-01T19:37:45.781802Z","iopub.status.idle":"2021-08-01T19:38:54.734418Z","shell.execute_reply.started":"2021-08-01T19:37:45.781773Z","shell.execute_reply":"2021-08-01T19:38:54.733456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-01T19:38:54.737493Z","iopub.execute_input":"2021-08-01T19:38:54.737851Z","iopub.status.idle":"2021-08-01T19:38:54.741905Z","shell.execute_reply.started":"2021-08-01T19:38:54.737808Z","shell.execute_reply":"2021-08-01T19:38:54.740999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nif df.shape[0] == 2479:\n#if df.shape[0] == 2477: #orig\n    fast_sub = True\n    print('fast')\n    fast_df = pd.DataFrame(([['00086460a852_study', 'negative 1 0 0 1 1'], \n                         ['000c9c05fd14_study', 'negative 1 0 0 1 1'], \n                         ['65761e66de9f_image', 'none 1 0 0 1 1'], \n                         ['51759b5579bc_image', 'none 1 0 0 1 1']]), \n                       columns=['id', 'PredictionString'])\nelse:\n    fast_sub = False\n    print('slow')\n    \n","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:54.74547Z","iopub.execute_input":"2021-08-01T19:38:54.746014Z","iopub.status.idle":"2021-08-01T19:38:54.784898Z","shell.execute_reply.started":"2021-08-01T19:38:54.745975Z","shell.execute_reply":"2021-08-01T19:38:54.783913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# .dcm to .png","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:54.789466Z","iopub.execute_input":"2021-08-01T19:38:54.791256Z","iopub.status.idle":"2021-08-01T19:38:55.140826Z","shell.execute_reply.started":"2021-08-01T19:38:54.79121Z","shell.execute_reply":"2021-08-01T19:38:55.139792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:55.145293Z","iopub.execute_input":"2021-08-01T19:38:55.147568Z","iopub.status.idle":"2021-08-01T19:38:55.157242Z","shell.execute_reply.started":"2021-08-01T19:38:55.147516Z","shell.execute_reply":"2021-08-01T19:38:55.153956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsplit = 'test'\nsave_dir = f'/kaggle/tmp/{split}/'\n\nos.makedirs(save_dir, exist_ok=True)\n\nsave_dir = f'/kaggle/tmp/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=600)  \n    study = '00086460a852' + '_study.png'\n    im.save(os.path.join(save_dir, study))\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=600)  \n    study = '000c9c05fd14' + '_study.png'\n    im.save(os.path.join(save_dir, study))\nelse:   \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=600)\n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))\n","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:55.161313Z","iopub.execute_input":"2021-08-01T19:38:55.164019Z","iopub.status.idle":"2021-08-01T19:38:56.240407Z","shell.execute_reply.started":"2021-08-01T19:38:55.16398Z","shell.execute_reply":"2021-08-01T19:38:56.239557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\nsave_dir = f'/kaggle/tmp/{split}/image/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size=512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=512)  \n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)\nmeta = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:56.241772Z","iopub.execute_input":"2021-08-01T19:38:56.242116Z","iopub.status.idle":"2021-08-01T19:38:56.651544Z","shell.execute_reply.started":"2021-08-01T19:38:56.242066Z","shell.execute_reply":"2021-08-01T19:38:56.650671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study predict","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nif fast_sub:\n    df = fast_df.copy()\nelse:\n    df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\n\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:56.655399Z","iopub.execute_input":"2021-08-01T19:38:56.655682Z","iopub.status.idle":"2021-08-01T19:38:56.672147Z","shell.execute_reply.started":"2021-08-01T19:38:56.655652Z","shell.execute_reply":"2021-08-01T19:38:56.671262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\n# Build model\nhub_url = '/kaggle/input/efficientnetv2-tf-hub/efficientnetv2-l-21k-ft1k/feature-vector'\nimage_size = 600","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:38:56.67604Z","iopub.execute_input":"2021-08-01T19:38:56.676381Z","iopub.status.idle":"2021-08-01T19:39:00.742118Z","shell.execute_reply.started":"2021-08-01T19:38:56.676351Z","shell.execute_reply":"2021-08-01T19:39:00.741267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/kerasapplications -q\n!pip install /kaggle/input/efficientnet-keras-source-code/ -q --no-deps\n\nimport os\n\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n\n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n\n    def decode_with_labels(path, label):\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n\n    def augment_with_labels(img, label):\n        return augment(img), label\n\n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n\n    return dset\n\n#COMPETITION_NAME = \"siim-cov19-test-img512-study-600\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\n\nIMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512)\n\n#load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\nif fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[:study_len]\ntest_paths = f'/kaggle/tmp/{split}/study/' + sub_df['id'] +'.png'\n\nsub_df['negative'] = 0\nsub_df['typical'] = 0\nsub_df['indeterminate'] = 0\nsub_df['atypical'] = 0\n\n\nlabel_cols = sub_df.columns[2:]\n\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)\n\nimage_size = 600\nbatch_size = BATCH_SIZE\nlabels = [\"negative\", \"typical\", \"indeterminate\", \"atypical\"]\n\nwith strategy.scope():\n    \n    models = []\n    \n    #models0 = tf.keras.Sequential([\n    #        # Explicitly define the input shape so the model can be properly\n    #        # loaded by the TFLiteConverter\n    #        tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n    #            hub.KerasLayer(hub_url, trainable=False),\n    #        tf.keras.layers.Dropout(rate=0.2),\n    #        tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='softmax')\n    #    ])\n    #models0.load_weights('../input/effv2-covid19detection-kf-stratified/model0.h5')\n    #models1 = tf.keras.Sequential([\n    #        # Explicitly define the input shape so the model can be properly\n    #        # loaded by the TFLiteConverter\n    #        tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n    #            hub.KerasLayer(hub_url, trainable=False),\n    #        tf.keras.layers.Dropout(rate=0.2),\n    #        tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='softmax')\n    #    ])\n    #models1.load_weights('../input/effv2-covid19detection-kf-stratified/model1.h5')\n    #models2 = tf.keras.Sequential([\n    #        # Explicitly define the input shape so the model can be properly\n    #        # loaded by the TFLiteConverter\n    #        tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n    #            hub.KerasLayer(hub_url, trainable=False),\n    #        tf.keras.layers.Dropout(rate=0.2),\n    #        tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='softmax')\n    #    ])\n    #models2.load_weights('../input/effv2-covid19detection-kf-stratified/model2.h5')\n    #models3 = tf.keras.Sequential([\n    #        # Explicitly define the input shape so the model can be properly\n    #        # loaded by the TFLiteConverter\n    #        tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n    #            hub.KerasLayer(hub_url, trainable=False),\n    #        tf.keras.layers.Dropout(rate=0.2),\n    #        tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='softmax')\n    #    ])\n    #models3.load_weights('../input/effv2-covid19detection-kf-stratified/model3.h5')\n    #models4 = tf.keras.Sequential([\n    #        # Explicitly define the input shape so the model can be properly\n    #        # loaded by the TFLiteConverter\n    #        tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n    #            hub.KerasLayer(hub_url, trainable=False),\n    #        tf.keras.layers.Dropout(rate=0.2),\n    #        tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='softmax')\n    #    ])\n    #models4.load_weights('../input/effv2-covid19detection-kf-stratified/model4.h5')\n    \n    models5 = tf.keras.models.load_model(\n        '../input/b7600stra216m01/model0.h5'\n    )\n    models6 = tf.keras.models.load_model(\n        '../input/b7600stra216m01/model1.h5'\n    )\n    models7 = tf.keras.models.load_model(\n        '../input/b7600stra216m01/model2.h5'\n    )\n    models8 = tf.keras.models.load_model(\n        '../input/b7600stra216m01/model3.h5'\n    )\n    models9 = tf.keras.models.load_model(\n        '../input/b7600stra216m01/model4.h5'\n    )\n    models0 = tf.keras.models.load_model(\n        '../input/covid19detection-first-notebook/model0.h5'\n    )\n    models1 = tf.keras.models.load_model(\n        '../input/covid19detection-first-notebook/model1.h5'\n    )\n    models2 = tf.keras.models.load_model(\n        '../input/covid19detection-first-notebook/model2.h5'\n    )\n    models3 = tf.keras.models.load_model(\n        '../input/covid19detection-first-notebook/model3.h5'\n    )\n    \n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    #models.append(models4)\n    models.append(models5)\n    models.append(models6)\n    models.append(models7)\n    models.append(models8)\n    models.append(models9)\n\n    \n    \n    \nsub_df[label_cols] = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:39:00.743394Z","iopub.execute_input":"2021-08-01T19:39:00.743738Z","iopub.status.idle":"2021-08-01T19:43:28.404096Z","shell.execute_reply.started":"2021-08-01T19:39:00.743701Z","shell.execute_reply":"2021-08-01T19:43:28.403108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del models\ndel models0, models1, models2, models3, models5,models6,models7,models8,models9\n#del models0, models1, models2, models3, models4","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:43:28.405631Z","iopub.execute_input":"2021-08-01T19:43:28.406097Z","iopub.status.idle":"2021-08-01T19:43:28.414104Z","shell.execute_reply.started":"2021-08-01T19:43:28.406031Z","shell.execute_reply":"2021-08-01T19:43:28.413269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.columns = ['id', 'PredictionString1', 'negative', 'typical', 'indeterminate', 'atypical']\ndf = pd.merge(df, sub_df, on = 'id', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:43:28.415298Z","iopub.execute_input":"2021-08-01T19:43:28.415698Z","iopub.status.idle":"2021-08-01T19:43:28.433096Z","shell.execute_reply.started":"2021-08-01T19:43:28.415656Z","shell.execute_reply":"2021-08-01T19:43:28.432163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"for i in range(study_len):\n    negative = df.loc[i,'negative']\n    typical = df.loc[i,'typical']\n    indeterminate = df.loc[i,'indeterminate']\n    atypical = df.loc[i,'atypical']\n    df.loc[i, 'PredictionString'] = f'negative {negative} 0 0 1 1 typical {typical} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1'","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:43:28.434438Z","iopub.execute_input":"2021-08-01T19:43:28.434804Z","iopub.status.idle":"2021-08-01T19:43:28.44207Z","shell.execute_reply.started":"2021-08-01T19:43:28.434767Z","shell.execute_reply":"2021-08-01T19:43:28.440921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study = df[['id', 'PredictionString']]\n\n# df.to_csv('submission.csv',index=False)\n# df","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:43:28.443425Z","iopub.execute_input":"2021-08-01T19:43:28.443979Z","iopub.status.idle":"2021-08-01T19:43:28.456193Z","shell.execute_reply.started":"2021-08-01T19:43:28.443939Z","shell.execute_reply":"2021-08-01T19:43:28.455351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 class","metadata":{}},{"cell_type":"code","source":"if fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[study_len:]\ntest_paths = f'/kaggle/tmp/{split}/image/' + sub_df['id'] +'.png'\nsub_df['none'] = 0\n\nlabel_cols = sub_df.columns[2]\n\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)\n\nwith strategy.scope():\n    \n    models = []\n    \n    #models0 = tf.keras.models.load_model(\n    #    '../input/effv2l-covid19detection-kf-stratified-2ndclass/model0.h5'\n    #)\n    #models1 = tf.keras.models.load_model(\n    #    '../input/effv2l-covid19detection-kf-stratified-2ndclass/model1.h5'\n    #)\n    #models2 = tf.keras.models.load_model(\n    #    '../input/effv2l-covid19detection-kf-stratified-2ndclass/model2.h5'\n    #)\n    #models3 = tf.keras.models.load_model(\n    #    '../input/effv2l-covid19detection-kf-stratified-2ndclass/model3.h5'\n    #)\n    #models4 = tf.keras.models.load_model(\n    #    '../input/effv2l-covid19detection-kf-stratified-2ndclass/model4.h5'\n    #)\n    #models0 = tf.keras.Sequential([\n    #        # Explicitly define the input shape so the model can be properly\n    #        # loaded by the TFLiteConverter\n    #        tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n    #            hub.KerasLayer(hub_url, trainable=False),\n    #        tf.keras.layers.Dropout(rate=0.2),\n    #        tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='softmax')\n    #    ])\n    #models0.load_weights('../input/effv2-covid19detection-kf-stratified/model0.h5')\n    \n    models0 = tf.keras.Sequential([\n            # Explicitly define the input shape so the model can be properly\n            # loaded by the TFLiteConverter\n            tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n                hub.KerasLayer(hub_url, trainable=False),\n            tf.keras.layers.Dropout(rate=0.2),\n            tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='sigmoid')\n        ])\n    models0.load_weights('../input/effv221k-c19-kf-str-2ndcl/model0.h5')\n    models1 = tf.keras.Sequential([\n            # Explicitly define the input shape so the model can be properly\n            # loaded by the TFLiteConverter\n            tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n                hub.KerasLayer(hub_url, trainable=False),\n            tf.keras.layers.Dropout(rate=0.2),\n            tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='sigmoid')\n        ])\n    models1.load_weights('../input/effv221k-c19-kf-str-2ndcl/model1.h5')\n    models2 = tf.keras.Sequential([\n            # Explicitly define the input shape so the model can be properly\n            # loaded by the TFLiteConverter\n            tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n                hub.KerasLayer(hub_url, trainable=False),\n            tf.keras.layers.Dropout(rate=0.2),\n            tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='sigmoid')\n        ])\n    models2.load_weights('../input/effv221k-c19-kf-str-2ndcl/model2.h5')\n    models3 = tf.keras.Sequential([\n            # Explicitly define the input shape so the model can be properly\n            # loaded by the TFLiteConverter\n            tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n                hub.KerasLayer(hub_url, trainable=False),\n            tf.keras.layers.Dropout(rate=0.2),\n            tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='sigmoid')\n        ])\n    models3.load_weights('../input/effv221k-c19-kf-str-2ndcl/model3.h5')\n    models4 = tf.keras.Sequential([\n            # Explicitly define the input shape so the model can be properly\n            # loaded by the TFLiteConverter\n            tf.keras.layers.InputLayer(input_shape=[image_size, image_size, 3]),\n                hub.KerasLayer(hub_url, trainable=False),\n            tf.keras.layers.Dropout(rate=0.2),\n            tf.keras.layers.Dense(len(labels),kernel_regularizer=tf.keras.regularizers.l2(0.0001),activation='sigmoid')\n        ])\n    models4.load_weights('../input/effv221k-c19-kf-str-2ndcl/model4.h5')\n \n    \n    models.append(models0)\n    models.append(models1)\n    models.append(models2)\n    models.append(models3)\n    models.append(models4)\n\n    \n    \n    \nsub_df[label_cols] = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)\ndf_2class = sub_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:43:28.457535Z","iopub.execute_input":"2021-08-01T19:43:28.457885Z","iopub.status.idle":"2021-08-01T19:46:45.248882Z","shell.execute_reply.started":"2021-08-01T19:43:28.45785Z","shell.execute_reply":"2021-08-01T19:46:45.24803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del models\ndel models0, models1, models2, models3, models4","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:45.250218Z","iopub.execute_input":"2021-08-01T19:46:45.250522Z","iopub.status.idle":"2021-08-01T19:46:45.255515Z","shell.execute_reply.started":"2021-08-01T19:46:45.25049Z","shell.execute_reply":"2021-08-01T19:46:45.254496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numba import cuda\nimport torch\ncuda.select_device(0)\ncuda.close()\ncuda.select_device(0)","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:45.256868Z","iopub.execute_input":"2021-08-01T19:46:45.257245Z","iopub.status.idle":"2021-08-01T19:46:47.276413Z","shell.execute_reply.started":"2021-08-01T19:46:45.257207Z","shell.execute_reply":"2021-08-01T19:46:47.275598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# yolov5 predict","metadata":{}},{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nimport torch","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:47.277741Z","iopub.execute_input":"2021-08-01T19:46:47.278153Z","iopub.status.idle":"2021-08-01T19:46:47.283744Z","shell.execute_reply.started":"2021-08-01T19:46:47.278113Z","shell.execute_reply":"2021-08-01T19:46:47.282789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = meta[meta['split'] == 'test']\nif fast_sub:\n    test_df = fast_df.copy()\nelse:\n    test_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\ntest_df = df[study_len:].reset_index(drop=True) \nmeta['image_id'] = meta['image_id'] + '_image'\nmeta.columns = ['id', 'dim0', 'dim1', 'split']\ntest_df = pd.merge(test_df, meta, on = 'id', how = 'left')\n","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:47.285199Z","iopub.execute_input":"2021-08-01T19:46:47.285674Z","iopub.status.idle":"2021-08-01T19:46:47.300035Z","shell.execute_reply.started":"2021-08-01T19:46:47.285633Z","shell.execute_reply":"2021-08-01T19:46:47.299213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim = 512 #1024, 256, 'original'\ntest_dir = f'/kaggle/tmp/{split}/image'\nweights_dir = '/kaggle/input/siim-cov19-yolov5-train/yolov5/runs/train/exp/weights/best.pt'\n\nshutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\nos.chdir('/kaggle/working/yolov5') # install dependencies\n\nimport torch\n#from IPython.display import Image, clear_output  # to display images\n\n#clear_output()\n#print('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))\n_det_model_path='../input/covid19-det/'\npath_1=\"/kaggle/input/covid19-det/kaggle-siim-covid/exp/weights/best.pt\"\npath_2=\"/kaggle/input/covid19-det/kaggle-siim-covid/exp2/weights/best.pt\"\npath_3=\"/kaggle/input/covid19-det/kaggle-siim-covid/exp3/weights/best.pt\"\npath_4=\"/kaggle/input/covid19-det/kaggle-siim-covid/exp4/weights/best.pt\"\npath_5=\"/kaggle/input/covid19-det/kaggle-siim-covid/exp5/weights/best.pt\"\n\npath_6=  \"/kaggle/input/yoloxm01/kaggle-siim-covid/exp/weights/best.pt\"\npath_7=  \"/kaggle/input/covid19-det-x-m1/kaggle-siim-covid/exp/weights/best.pt\"\npath_8=  \"/kaggle/input/covid19-det-x-m2/kaggle-siim-covid/exp/weights/best.pt\"\npath_9=  \"/kaggle/input/covid19-det-x-m3/kaggle-siim-covid/exp/weights/best.pt\"\npath_10= \"/kaggle/input/covid19-det-x-m4/kaggle-siim-covid/exp/weights/best.pt\"\n\npath_11= \"/kaggle/input/yolov5lm02/kaggle-siim-covid/exp/weights/best.pt\"\npath_12= \"/kaggle/input/yolov5lm02/kaggle-siim-covid/exp2/weights/best.pt\"\npath_13= \"/kaggle/input/covid19-det-yolo-l-m4y5/kaggle-siim-covid/exp/weights/best.pt\"\npath_14= \"/kaggle/input/covid19-det-yolo-l-m4y5/kaggle-siim-covid/exp2/weights/best.pt\"\n\n\nMODEL_PATH = path_1 + \" \" + path_2 + \" \" + path_3 + \" \" + path_4 + \" \" + path_5 + \" \" + path_6 + \" \" + path_7 + \" \" + path_8 + \" \" + path_9 + \" \" + path_10 + \" \" + path_11 + \" \" + path_12 + \" \" + path_13 + \" \" + path_14 \nMODEL_PATH\n\n_test_files_path = \"/kaggle/input/covid19512/test/\"\nMODEL_PATH\n_data_dir = \"/kaggle/input/covid19512/\"\n\n\n!python detect.py --weights {MODEL_PATH} \\\n                  --source $test_dir\\\n                  --img 512 \\\n                  --conf 0.001 \\\n                  --iou-thres 0.5 \\\n                  --augment \\\n                  --save-txt \\\n                  --save-conf\n\n\ndef yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n\n    \"\"\" \n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n\n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n\n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n\n    return bboxes\n","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:47.301317Z","iopub.execute_input":"2021-08-01T19:46:47.301759Z","iopub.status.idle":"2021-08-01T19:46:58.255045Z","shell.execute_reply.started":"2021-08-01T19:46:47.301713Z","shell.execute_reply":"2021-08-01T19:46:58.25413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:58.256689Z","iopub.execute_input":"2021-08-01T19:46:58.257024Z","iopub.status.idle":"2021-08-01T19:46:58.278481Z","shell.execute_reply.started":"2021-08-01T19:46:58.256983Z","shell.execute_reply":"2021-08-01T19:46:58.277558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = []\nPredictionStrings = []\nmeta_df = pd.read_csv(_data_dir + \"meta.csv\")\n\n#for file_path in tqdm(glob('runs/detect/exp/labels/*.txt')):\nfor dir_path, _, filenames in os.walk(test_dir):\n        print(len(filenames))\nfor file in filenames:\n    file_path = 'runs/detect/exp/labels/' + file.replace(\".png\", '.txt')\n    \n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.id==image_id,['dim1', 'dim0']].values[0]\n    #w, h = meta_df.loc[meta_df.image_id == image_id,['dim1', 'dim0']].values[0]\n\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 12).astype(str))\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))\n\n\npred_df = pd.DataFrame({'id':image_ids,\n                        'PredictionString':PredictionStrings})","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:58.280086Z","iopub.execute_input":"2021-08-01T19:46:58.280474Z","iopub.status.idle":"2021-08-01T19:46:58.321225Z","shell.execute_reply.started":"2021-08-01T19:46:58.280436Z","shell.execute_reply":"2021-08-01T19:46:58.320382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(['PredictionString'], axis=1)\nsub_df = pd.merge(test_df, pred_df, on = 'id', how = 'left').fillna(\"none 1 0 0 1 1\")\nsub_df = sub_df[['id', 'PredictionString']]\nfor i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] == \"none 1 0 0 1 1\":\n        continue\n    sub_df_split = sub_df.loc[i,'PredictionString'].split()\n    sub_df_list = []\n    for j in range(int(len(sub_df_split) / 6)):\n        sub_df_list.append('opacity')\n        sub_df_list.append(sub_df_split[6 * j + 1])\n        sub_df_list.append(sub_df_split[6 * j + 2])\n        sub_df_list.append(sub_df_split[6 * j + 3])\n        sub_df_list.append(sub_df_split[6 * j + 4])\n        sub_df_list.append(sub_df_split[6 * j + 5])\n    sub_df.loc[i,'PredictionString'] = ' '.join(sub_df_list)\nsub_df['none'] = df_2class['none'] \nfor i in range(sub_df.shape[0]):\n    if sub_df.loc[i,'PredictionString'] != 'none 1 0 0 1 1':\n        sub_df.loc[i,'PredictionString'] = sub_df.loc[i,'PredictionString'] + ' none ' + str(sub_df.loc[i,'none']) + ' 0 0 1 1'\nsub_df = sub_df[['id', 'PredictionString']]   \ndf_study = df_study[:study_len]\ndf_study = df_study.append(sub_df).reset_index(drop=True)\ndf_study.to_csv('/kaggle/working/submission.csv',index = False)  \nshutil.rmtree('/kaggle/working/yolov5')","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:46:58.324491Z","iopub.execute_input":"2021-08-01T19:46:58.324756Z","iopub.status.idle":"2021-08-01T19:46:58.61616Z","shell.execute_reply.started":"2021-08-01T19:46:58.324715Z","shell.execute_reply":"2021-08-01T19:46:58.615373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_study","metadata":{"execution":{"iopub.status.busy":"2021-08-01T19:48:29.316819Z","iopub.execute_input":"2021-08-01T19:48:29.317246Z","iopub.status.idle":"2021-08-01T19:48:29.32938Z","shell.execute_reply.started":"2021-08-01T19:48:29.317211Z","shell.execute_reply":"2021-08-01T19:48:29.32838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}