{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"thanks to https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px  \nthanks to https://www.kaggle.com/awsaf49/vinbigdata-cxr-ad-yolov5-14-class-infer  \ntrain_study: https://www.kaggle.com/h053473666/siim-covid19-efnb7-train-study  \ntrain_image: https://www.kaggle.com/h053473666/siim-cov19-yolov5-train  \ntrain_2class: https://www.kaggle.com/h053473666/siim-covid19-efnb7-train-fold0-5-2class  \n  \nversion1:Original hyperparameters (yolov5)  \nversion4:New hyperparameters (yolov5)\n","metadata":{}},{"cell_type":"code","source":"match_mode, study_mode, image_mode = 1, 1, 1                # 正式上传两边的成绩\n#match_mode, study_mode, image_mode = 0, 1, 0  # 测试study label部分的预测能力\n#match_mode, study_mode, image_mode = 0, 0, 1   # 测试image部分的预测能力\n","metadata":{"execution":{"iopub.status.busy":"2021-07-12T00:09:52.476368Z","iopub.execute_input":"2021-07-12T00:09:52.47675Z","iopub.status.idle":"2021-07-12T00:09:52.485142Z","shell.execute_reply.started":"2021-07-12T00:09:52.476669Z","shell.execute_reply":"2021-07-12T00:09:52.484329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-12T00:09:54.662186Z","iopub.execute_input":"2021-07-12T00:09:54.662497Z","iopub.status.idle":"2021-07-12T00:11:03.128267Z","shell.execute_reply.started":"2021-07-12T00:09:54.662468Z","shell.execute_reply":"2021-07-12T00:11:03.127347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-12T00:11:07.663287Z","iopub.execute_input":"2021-07-12T00:11:07.663606Z","iopub.status.idle":"2021-07-12T00:11:07.921589Z","shell.execute_reply.started":"2021-07-12T00:11:07.663574Z","shell.execute_reply":"2021-07-12T00:11:07.920786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nif df.shape[0] == 2477:\n    fast_sub = True\n    fast_df = pd.DataFrame(([['00086460a852_study', 'negative 1 0 0 1 1'], \n                         ['000c9c05fd14_study', 'negative 1 0 0 1 1'], \n                         ['65761e66de9f_image', 'none 1 0 0 1 1'], \n                         ['51759b5579bc_image', 'none 1 0 0 1 1']]), \n                       columns=['id', 'PredictionString'])\nelse:\n    fast_sub = False\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data   \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    im = Image.fromarray(array)  \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)    \n    return im\nsplit = 'test'\nsave_dir = f'/kaggle/tmp/{split}/'\nos.makedirs(save_dir, exist_ok=True)\nsave_dir = f'/kaggle/tmp/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size= 600)  \n    study = '00086460a852' + '_study.png'\n    im.save(os.path.join(save_dir, study))\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size = 600)  \n    study = '000c9c05fd14' + '_study.png'\n    im.save(os.path.join(save_dir, study))\nelse:   \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size = 600)  \n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))\nimage_id = []\ndim0 = []\ndim1 = []\nsplits = []\nsave_dir = f'/kaggle/tmp/{split}/image/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub:\n    xray = read_xray('../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm')\n    im = resize(xray, size = 512) # original 512 \n    im.save(os.path.join(save_dir,'65761e66de9f_image.png'))\n    image_id.append('65761e66de9f.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    xray = read_xray('../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm')\n    im = resize(xray, size = 512)  \n    im.save(os.path.join(save_dir, '51759b5579bc_image.png'))\n    image_id.append('51759b5579bc.dcm'.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\nelse:\n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size = 512)   # original 512 \n            im.save(os.path.join(save_dir, file.replace('.dcm', '_image.png')))\n            image_id.append(file.replace('.dcm', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])\n            splits.append(split)\nmeta = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})\nimport numpy as np \nimport pandas as pd\nif fast_sub:\n    df = fast_df.copy()\nelse:\n    df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nid_laststr_list  = []\nfor i in range(df.shape[0]):\n    id_laststr_list.append(df.loc[i,'id'][-1])\ndf['id_last_str'] = id_laststr_list\nstudy_len = df[df['id_last_str'] == 'y'].shape[0]\n!pip install /kaggle/input/kerasapplications -q\n!pip install /kaggle/input/efficientnet-keras-source-code/ -q --no-deps\nimport efficientnet.tfkeras as efn\nimport tensorflow as tf\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    return strategy\ndef build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n        img = tf.cast(img, tf.float32) / 255.0   \n        img = tf.image.resize(img, target_size)\n        return img\n    def decode_with_labels(path, label):\n        return decode(path), label\n    return decode_with_labels if with_labels else decode\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    def augment_with_labels(img, label):\n        return augment(img), label\n    return augment_with_labels if with_labels else augment\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    return dset\n#COMPETITION_NAME = \"siim-cov19-test-img512-study-600\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 28\nIMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512, 704)\n#load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\nif fast_sub:\n    sub_df = fast_df.copy()\nelse:\n    sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nsub_df = sub_df[:study_len]\ntest_paths = f'/kaggle/tmp/{split}/study/' + sub_df['id'] +'.png'\nsub_df['negative'] = 0\nsub_df['typical'] = 0\nsub_df['indeterminate'] = 0\nsub_df['atypical'] = 0\nlabel_cols = sub_df.columns[2:]\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False,\n    decode_fn=test_decoder\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T00:11:10.944238Z","iopub.execute_input":"2021-07-12T00:11:10.944551Z","iopub.status.idle":"2021-07-12T00:12:10.654476Z","shell.execute_reply.started":"2021-07-12T00:11:10.944522Z","shell.execute_reply":"2021-07-12T00:12:10.653626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if study_mode:  \n    with strategy.scope():\n\n    \n#         10 models max\n#      m1 - m7   reach 0.622- >   lb 0.453\n# # mean validation score 0.5043\n#         m1 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-study/model1.h5') # lb 0.428 'mAP2 0.358'\n#         m2 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-study/model4.h5') # lb 0.433 'mAP2 0.3735'\n#         m3 = tf.keras.models.load_model('../input/616study/model0.h5') # lb 0.439 'mAP 0.357'\n#         m75 = tf.keras.models.load_model('../input/731kushalmodel1/model1.h5') # lb 0.429 'mAP 0.3691'\n#         #m4 = tf.keras.models.load_model('../input/616study/model1.h5') # lb 0.426 mAP 0.364\n#         m5 = tf.keras.models.load_model('../input/616study/model2.h5') # lb 0.435 mAP 0.378\n#         m6 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model2.h5') # lb 0.434 mAP 0.383\n#         m7 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model3.h5') # lb 0.435 mAP 0.37\n#         m9 = tf.keras.models.load_model('../input/77-study-seed-27/model2.h5') # lb 0.428  mAP 0.38 \n#         m43 = tf.keras.models.load_model('../input/724-study/model4.h5') # lb 0.429 'mAP 0.373'\n#         m32 = tf.keras.models.load_model('../input/716-study-extra/model4.h5') # lb 0.429 'mAP 0.383'\n#         models = [m1, m2, m3, m75, m5, m6, m7, m9, m43, m32] \n# #  ------------------------------------ 0.621 above ----------------------------------------------\n# # high CV balanced -----------------------                mAP2 > 0.36\n# # mean validation score 0.5130528221799316\n#         m10 = tf.keras.models.load_model('../input/710-study-brightness-random-saturation-auc-pr/model0.h5') # lb 0.413 mAP 0.362\n#         m74 =  tf.keras.models.load_model('../input/731-study/model0.h5') # lb 0.420 'mAP 0.361\n#         m29 = tf.keras.models.load_model('../input/716-study-extra/model1.h5') # lb 0.426 'mAP 0.365'\n#         m75 = tf.keras.models.load_model('../input/731kushalmodel1/model1.h5') # lb 0.429 'mAP 0.3691'\n#         m6 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model2.h5') # lb 0.434 mAP 0.383\n#         m9 = tf.keras.models.load_model('../input/77-study-seed-27/model2.h5') # lb 0.428  mAP 0.38 \n#         m8 = tf.keras.models.load_model('../input/616study/model3.h5') # lb 0.417 mAP 0.372\n#         m7 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model3.h5') # lb 0.435 mAP 0.37\n#         m32 = tf.keras.models.load_model('../input/716-study-extra/model4.h5') # lb 0.429 'mAP 0.383'\n#         m68 = tf.keras.models.load_model('../input/724-study/model4.h5') # lb 0.429 'mAP 0.3808'\n        \n#         models = [m10, m74, m29, m75, m6, m9, m8, m7, m32, m68]\n# # high CV balanced-----------------------  \n\n# # high CV mixed with high lb-----------------------  \n        m3 = tf.keras.models.load_model('../input/616study/model0.h5') # lb 0.439 'mAP 0.357'\n        m10 = tf.keras.models.load_model('../input/710-study-brightness-random-saturation-auc-pr/model0.h5') # lb 0.413 mAP 0.362\n        m75 = tf.keras.models.load_model('../input/731kushalmodel1/model1.h5') # lb 0.429 'mAP 0.3691'\n        m1 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-study/model1.h5') # lb 0.428 'mAP2 0.358'\n        m6 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model2.h5') # lb 0.434 mAP 0.383\n        m5 = tf.keras.models.load_model('../input/616study/model2.h5') # lb 0.435 mAP 0.378\n        m8 = tf.keras.models.load_model('../input/616study/model3.h5') # lb 0.417 mAP 0.372\n        m7 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model3.h5') # lb 0.435 mAP 0.37\n        m2 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-study/model4.h5') # lb 0.433 'mAP2 0.3735'\n        m32 = tf.keras.models.load_model('../input/716-study-extra/model4.h5') # lb 0.429 'mAP 0.383'\n        \n        models = [m3, m10, m75, m1, m6, m5, m8, m7, m2, m32]\n        \n# # high CV mixed with high lb-----------------------  \n#         m6 = tf.keras.models.load_model('../input/77-noisy-student-fine-tuning/model2.h5') # lb 0.434 mAP2 0.383\n#         m8 = tf.keras.models.load_model('../input/616study/model3.h5') # lb 0.417 mAP 0.372\n#         m9 = tf.keras.models.load_model('../input/77-study-seed-27/model2.h5') # lb 0.428  mAP 0.38    \n        \n#         m11 = tf.keras.models.load_model('../input/710-study-brightness-random-saturation-auc-pr/model2.h5') # lb 0.423 mAP 0.375\n#         m12 = tf.keras.models.load_model('../input/710-study-brightness-random-saturation-auc-pr/model3.h5') # lb  mAP 0.359\n#         m13 = tf.keras.models.load_model('../input/710-study-brightness-random-saturation-auc-pr/model4.h5') # lb  mAP 0.362\n# # # extra data + brightness, auc PR, ---------------------------------------------------------------------------------\n#         m14 = tf.keras.models.load_model('../input/713-seed-77-bright-satura/model1.h5') # mAP 0.349\n#         m15 = tf.keras.models.load_model('../input/713-seed-77-bright-satura/model2.h5') # mAP 0.362\n#         m16 = tf.keras.models.load_model('../input/713-seed-77-bright-satura/model3.h5') # mAP 0.349\n#         m17 = tf.keras.models.load_model('../input/713-seed-77-bright-satura/model4.h5') # lb 0.406 mAP 0.372\n# # # # extra data + contrast, auc PR  summation_method = 'majoring' ----------------------------------------------------\n#         m18 = tf.keras.models.load_model('../input/713-auc-pr-majoring-contrast/model0.h5') # 'mAP 0.349' \n#         m19 = tf.keras.models.load_model('../input/713-auc-pr-majoring-contrast/model1.h5') # 'mAP 0.346'\n#         m20 = tf.keras.models.load_model('../input/713-auc-pr-majoring-contrast/model2.h5') # lb 0.408 'mAP 0.366'\n#         m21 = tf.keras.models.load_model('../input/713-auc-pr-majoring-contrast/model3.h5') # 'mAP 0.346'\n#         m22 = tf.keras.models.load_model('../input/713-auc-pr-majoring-contrast/model4.h5') # lb 0.410 'mAP 0.372'\n\n# # # # # no extra data + h flip, auc PR  summation_method = 'majoring'\n#         m23 = tf.keras.models.load_model('../input/713-h-ud-flip-auc-pr/model0.h5') # 'mAP 0.356'\n#         m24 = tf.keras.models.load_model('../input/713-h-ud-flip-auc-pr/model1.h5') # 'mAP 0.349'\n#         m25 = tf.keras.models.load_model('../input/713-h-ud-flip-auc-pr/model2.h5') # 'mAP 0.375'\n#         m26 = tf.keras.models.load_model('../input/713-h-ud-flip-auc-pr/model3.h5') # 'mAP 0.355'\n#         m27 = tf.keras.models.load_model('../input/713-h-ud-flip-auc-pr/model4.h5') # lb 0.426 'mAP 0.38'\n\n# ### extra data \n#         m28 = tf.keras.models.load_model('../input/716-study-extra/model0.h5') # 'mAP 0.35988'\n#         m29 = tf.keras.models.load_model('../input/716-study-extra/model1.h5') # lb 0.426 'mAP 0.365'\n#         m30 = tf.keras.models.load_model('../input/716-study-extra/model2.h5') # lb 0.429 'mAP 0.373'\n#         m31 = tf.keras.models.load_model('../input/716-study-extra/model3.h5') # lb 0.422'mAP 0.363'\n#         m32 = tf.keras.models.load_model('../input/716-study-extra/model4.h5') # lb 0.429 'mAP 0.383'\n\n# ## Kushal\n#         m33 = tf.keras.models.load_model('../input/study-noisystudentep38-seed11-folds5/model0.h5') # mAP 0.343'\n#         m34 = tf.keras.models.load_model('../input/study-noisystudentep38-seed11-folds5/model1.h5') # 'mAP2 0.358'\n#         m35 = tf.keras.models.load_model('../input/study-noisystudentep38-seed11-folds5/model2.h5') # lb 0.428 'mAP2 0.368'\n#         m36 = tf.keras.models.load_model('../input/study-noisystudentep38-seed11-folds5/model3.h5') # lb 0.425 'mAP2 0.333'\n#         m37 = tf.keras.models.load_model('../input/study-noisystudentep38-seed11-folds5/model4.h5') # lb 0.420 'mAP2 0.366'\n        \n# ## Kushal\n#         m39 = tf.keras.models.load_model('../input/724-study/model0.h5') # 'mAP 0.346'\n#         m40 = tf.keras.models.load_model('../input/724-study/model1.h5') # 'mAP 0.356'\n#         m41 = tf.keras.models.load_model('../input/724-study/model2.h5') # lb 0.412 'mAP 0.377'\n#         m42 = tf.keras.models.load_model('../input/724-study/model3.h5') # 'mAP 0.351'\n#         m43 = tf.keras.models.load_model('../input/724-study/model4.h5') # lb 0.429 'mAP 0.373'\n      \n#         m44 = tf.keras.models.load_model('../input/study-noisystudentep25-seed101-folds5/model0.h5') # 'mAP 0.345'\n#         m45 = tf.keras.models.load_model('../input/study-noisystudentep25-seed101-folds5/model1.h5') # 'mAP 0.362'\n#         m46 = tf.keras.models.load_model('../input/study-noisystudentep25-seed101-folds5/model2.h5') # 'mAP 0.369'\n#         m47 = tf.keras.models.load_model('../input/study-noisystudentep25-seed101-folds5/model3.h5') # lb 0.419 'mAP 0.359'\n#         m48 = tf.keras.models.load_model('../input/study-noisystudentep25-seed101-folds5/model4.h5') # lb 0.414 'mAP 0.374'\n\n#         m49 = tf.keras.models.load_model('../input/study-noisystudentep30-seed241-folds5/model0.h5') #  'mAP 0.344'\n#         m50 = tf.keras.models.load_model('../input/study-noisystudentep30-seed241-folds5/model1.h5') #  'mAP 0.363'\n#         m51 = tf.keras.models.load_model('../input/study-noisystudentep30-seed241-folds5/model2.h5') #  'mAP 0.367'\n#         m52 = tf.keras.models.load_model('../input/study-noisystudentep30-seed241-folds5/model3.h5') #  'mAP 0.343'\n#         m53 = tf.keras.models.load_model('../input/study-noisystudentep30-seed241-folds5/model4.h5') # lb 0.405 mAP2 0.355'\n#         # v2\n#         m54 = tf.keras.models.load_model('../input/study-efficientnetv2l21kft1k-ep38-sd101/model0.h5') # 'mAP 0.3319'\n#         m55 = tf.keras.models.load_model('../input/study-efficientnetv2l21kft1k-ep38-sd101/model1.h5') # 'mAP 0.3334'\n#         m56 = tf.keras.models.load_model('../input/study-efficientnetv2l21kft1k-ep38-sd101/model2.h5') # 'mAP 0.3551'\n#         m57 = tf.keras.models.load_model('../input/study-efficientnetv2l21kft1k-ep38-sd101/model3.h5') # 'mAP 0.3406'\n#         m58 = tf.keras.models.load_model('../input/study-efficientnetv2l21kft1k-ep38-sd101/model4.h5') # 'mAP 0.3669'    \n#         # v2\n#         m59 = tf.keras.models.load_model('../input/study-ns-ep38-seed101/model0.h5') # 'mAP 0.3424'\n#         m60 = tf.keras.models.load_model('../input/study-ns-ep38-seed101/model1.h5') # 'mAP 0.3471'\n#         m61 = tf.keras.models.load_model('../input/study-ns-ep38-seed101/model2.h5') # 'mAP 0.359'\n#         m62 = tf.keras.models.load_model('../input/study-ns-ep38-seed101/model3.h5') # 'mAP 0.3567'\n#         m63 = tf.keras.models.load_model('../input/study-ns-ep38-seed101/model4.h5') # lb 0.408 'mAP 0.3753' \n\n\n#         m64 = tf.keras.models.load_model('../input/724-study/model0.h5') # 'mAP 0.3522'\n#         m65 = tf.keras.models.load_model('../input/724-study/model1.h5') # 'mAP 0.3364'\n#         m66 = tf.keras.models.load_model('../input/724-study/model2.h5') # 'mAP 0.3594'\n#         m67 = tf.keras.models.load_model('../input/724-study/model3.h5') # 'mAP 0.3506'\n#         m68 = tf.keras.models.load_model('../input/724-study/model4.h5') # lb 0.429 'mAP 0.3808'      \n\n#         m69 = tf.keras.models.load_model('../input/725-study-no-e/model0.h5') # 'mAP 0.3508'\n#         m70 = tf.keras.models.load_model('../input/725-study-no-e/model1.h5') # 'mAP 0.3565'\n#         m71 = tf.keras.models.load_model('../input/725-study-no-e/model2.h5') # 'mAP 0.3688'\n#         m72 = tf.keras.models.load_model('../input/725-study-no-e/model3.h5') # lb 0.429 'mAP 0.3624'\n#         m73 = tf.keras.models.load_model('../input/725-study-no-e/model4.h5') # lb 0.414'mAP 0.3671' \n\n        sub_df[label_cols] = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)\n        sub_df.columns = ['id', 'PredictionString1', 'negative', 'typical', 'indeterminate', 'atypical']\n        df = pd.merge(df, sub_df, on = 'id', how = 'left')\nif (not match_mode) and image_mode: \n    with strategy.scope():\n        model = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-study/model4.h5') # lb 0.433\n        models = [model]\n    sub_df[label_cols] = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)\n    sub_df.columns = ['id', 'PredictionString1', 'negative', 'typical', 'indeterminate', 'atypical']\n    df = pd.merge(df, sub_df, on = 'id', how = 'left')  ","metadata":{"execution":{"iopub.status.busy":"2021-07-27T12:23:13.79372Z","iopub.execute_input":"2021-07-27T12:23:13.794412Z","iopub.status.idle":"2021-07-27T12:23:13.878999Z","shell.execute_reply.started":"2021-07-27T12:23:13.794276Z","shell.execute_reply":"2021-07-27T12:23:13.877864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# study string","metadata":{}},{"cell_type":"code","source":"for i in range(study_len):\n    negative = df.loc[i,'negative']\n    typical = df.loc[i,'typical']\n    indeterminate = df.loc[i,'indeterminate']\n    atypical = df.loc[i,'atypical']\n    df.loc[i, 'PredictionString'] = f'negative {negative} 0 0 1 1 indeterminate {indeterminate} 0 0 1 1 atypical {atypical} 0 0 1 1 typical {typical} 0 0 1 1'\ndf_study = df[['id', 'PredictionString']]\nif not image_mode:\n    df_study.to_csv('submission.csv',index=False)\n    df_study","metadata":{"execution":{"iopub.status.busy":"2021-07-12T00:13:08.636882Z","iopub.execute_input":"2021-07-12T00:13:08.637207Z","iopub.status.idle":"2021-07-12T00:13:08.64617Z","shell.execute_reply.started":"2021-07-12T00:13:08.637176Z","shell.execute_reply":"2021-07-12T00:13:08.645126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 class","metadata":{}},{"cell_type":"code","source":"if image_mode:\n    if fast_sub:\n        sub_df = fast_df.copy()\n    else:\n        sub_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n    sub_df = sub_df[study_len:]\n    test_paths = f'/kaggle/tmp/{split}/image/' + sub_df['id'] +'.png'\n    sub_df['none'] = 0\n    label_cols = sub_df.columns[2]\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[8], IMSIZE[8]), ext='png')\n    dtest = build_dataset(\n        test_paths, bsize=BATCH_SIZE, repeat=False, \n        shuffle=False, augment=False, cache=False,\n        decode_fn=test_decoder\n    )\n    with strategy.scope():\n#         m12 = tf.keras.models.load_model('../input/7-15-two-class-some-augmentor/model2.h5')\n#         models1 = [m12]\n#         m0 = tf.keras.models.load_model('../input/624class2train/model0.h5') # lb 0.589  mAP 0.255\n#         m1 = tf.keras.models.load_model( '../input/siim-covid19-efnb7-train-fold0-5-2class/model1.h5') # lb 0.593  mAP 0.265\n#         m2 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model2.h5') # lb 0.589  mAP 0.252\n#         m3 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model3.h5') # lb 0.592  mAP 0.251\n#         m4 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model4.h5')  # lb 0.590  mAP 0.260   \n#         models = [m0, m1, m2, m3, m4]\n  # ---------------------------------------------------------0.622 above\n#     #  High CV score ----------------------------------------  # lb 0.584  -> mAP 0.257\n        m5 = tf.keras.models.load_model('../input/722-2-class-2/model0.h5') # lb 0.588 'mAP 0.26' \n        m11 = tf.keras.models.load_model('../input/7-15-two-class-some-augmentor/model1.h5') # lb 0.588 mAP 0.265\n        m12 = tf.keras.models.load_model('../input/7-15-two-class-some-augmentor/model2.h5') # lb 0.587 mAP 0.259\n        m16 = tf.keras.models.load_model('../input/722-2-class/model3.h5') # mAP 0.255\n        m9 = tf.keras.models.load_model('../input/721-2-class/model4.h5') # mAP 0.264\n        models = [m16, m5, m9, m11, m12]\n#  High CV score --------------------------------------\n# ---------------------------------------------------------\n#         m6 = tf.keras.models.load_model('../input/624class2train/model1.h5') # lb 0.591  mAP 0.264\n#         m7 = tf.keras.models.load_model('../input/624class2train/model2.h5') # lb 0.584  mAP 0.241\n#         m8 = tf.keras.models.load_model('../input/624class2train/model3.h5') # lb 0.587  mAP 0.244\n#         m10 = tf.keras.models.load_model('../input/7-15-two-class-some-augmentor/model0.h5') # lb 0.589 mAP 0.257\n# ---------------------------------------------------------\n#         m13 = tf.keras.models.load_model('../input/716-2-class/model0.h5') # mAP 0.256\n#         m14 = tf.keras.models.load_model('../input/716-2-class/model1.h5') # mAP 0.251\n#         m15 = tf.keras.models.load_model('../input/716-2-class/model2.h5') # mAP 0.249\n#         m17 = tf.keras.models.load_model('../input/716-2-class/model4.h5') # mAP 0.26\n##  check yolo score\n#         m0 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model0.h5') \n#         m1 = tf.keras.models.load_model( '../input/siim-covid19-efnb7-train-fold0-5-2class/model1.h5') \n#         m2 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model2.h5')\n#         m3 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model3.h5')\n#         m4 = tf.keras.models.load_model('../input/siim-covid19-efnb7-train-fold0-5-2class/model4.h5') \n#         models = [m0, m1, m2, m3, m4]\n##  check yolo score\n        pred1 = sum([model.predict(dtest, verbose=1) for model in models]) / len(models)\n\n        sub_df[label_cols] = pred1\n        df_2class = sub_df.reset_index(drop=True)\n    #     del m0, m1, m2, m3, m4\n    from numba import cuda\n    cuda.select_device(0)\n    cuda.close()\n    cuda.select_device(0)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T00:13:10.727796Z","iopub.execute_input":"2021-07-12T00:13:10.728155Z","iopub.status.idle":"2021-07-12T00:15:09.189079Z","shell.execute_reply.started":"2021-07-12T00:13:10.728126Z","shell.execute_reply":"2021-07-12T00:15:09.188172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# yolov5 predict","metadata":{}},{"cell_type":"code","source":"if image_mode:\n    import numpy as np, pandas as pd\n    from glob import glob\n    import shutil, os\n    import matplotlib.pyplot as plt\n    from sklearn.model_selection import GroupKFold\n    from tqdm.notebook import tqdm\n    import seaborn as sns\n    import torch\n    meta = meta[meta['split'] == 'test']\n    if fast_sub:\n        test_df = fast_df.copy()\n    else:\n        test_df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\n    test_df = df[study_len:].reset_index(drop=True) \n    meta['image_id'] = meta['image_id'] + '_image'\n    meta.columns = ['id', 'dim0', 'dim1', 'split']\n    test_df = pd.merge(test_df, meta, on = 'id', how = 'left')\n    dim = 672  # 672 \n    test_dir = f'/kaggle/tmp/{split}/image'\n    w1 = '/kaggle/input/619yolo-img-672-batch-16-epochs-19/yolov5/runs/train/exp/weights/best.pt' # lb 0.600\n    w2 = '/kaggle/input/74-yolo-train-image-672-epochs-20-folds-3/yolov5/runs/train/exp/weights/best.pt'# lb 0.600\n    w3 = '/kaggle/input/75-yolo-train-image-672-epochs-24-folds-5/yolov5/runs/train/exp/weights/best.pt'# lb 0.600\n    w4 = '/kaggle/input/75yolo-train-image672-epochs30-splits5-fold0/yolov5/runs/train/exp/weights/best.pt'# lb 0.600\n    w5 = '/kaggle/input/75yolo-train-image672-epochs30-splits5-folds1/yolov5/runs/train/exp/weights/best.pt'# lb 0.600\n    w6 = '/kaggle/input/75yolo-train-image672-epochs30-splits5-folds2/yolov5/runs/train/exp/weights/best.pt'# lb 0.602\n    w7 = '/kaggle/input/711-yolo-train-image704-epochs30-split5-fold4/yolov5/runs/train/exp/weights/best.pt' # lb 0.600\n    w8 = '/kaggle/input/715-yolo-train-img672-eph30-fld3-splt4-b8/yolov5/runs/train/exp/weights/best.pt' # lb 0.602\n    w9 = '/kaggle/input/718-yolo-train-img672-eph30-fld0-splt4-b8/yolov5/runs/train/exp/weights/best.pt' # lb 0.602\n    w10 = '/kaggle/input/718-yolo-train-img672-eph30-fld1-splt4-b8/yolov5/runs/train/exp/weights/best.pt' # lb 0.601\n    w11 = '/kaggle/input/718-yolo-train-img672-eph30-fld2-splt4-b8/yolov5/runs/train/exp/weights/best.pt' # lb 0.602\n    \n    shutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\n    os.chdir('/kaggle/working/yolov5') # install dependencies\n    import torch\n#  # when testing 2 class models\n#     !python detect.py --weights $w1\\\n#     --img 672\\\n#     --conf 0.001\\\n#     --iou 0.5\\\n#     --source $test_dir\\\n#     --save-txt --save-conf --exist-ok\n    # $w1 $w2 $w3 $w4 $w5 $w6 $w7 $w8 $w9\n    !python detect.py --weights $w1 $w2 $w3 $w4 $w5 $w6 $w7 $w8 $w9 $w10 $w11\\\n    --img 672\\\n    --conf 0.0001\\\n    --iou 0.5\\\n    --augment\\\n    --source $test_dir\\\n    --save-txt --save-conf --exist-ok\n    def yolo2voc(image_height, image_width, bboxes):\n        \"\"\"\n        yolo => [xmid, ymid, w, h] (normalized)\n        voc  => [x1, y1, x2, y1]\n        \"\"\" \n        bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n        bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n        bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n        bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n        bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n        return bboxes\n    image_ids = []\n    PredictionStrings = []\n    for file_path in tqdm(glob('runs/detect/exp/labels/*.txt')):\n        image_id = file_path.split('/')[-1].split('.')[0]\n        w, h = test_df.loc[test_df.id==image_id,['dim1', 'dim0']].values[0]\n        f = open(file_path, 'r')\n        data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n        data = data[:, [0, 5, 1, 2, 3, 4]]\n        bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 12).astype(str))\n        for idx in range(len(bboxes)):\n            bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n        image_ids.append(image_id)\n        PredictionStrings.append(' '.join(bboxes))\n    pred_df = pd.DataFrame({'id':image_ids,\n                            'PredictionString':PredictionStrings})\n    test_df = test_df.drop(['PredictionString'], axis=1)\n    sub_df = pd.merge(test_df, pred_df, on = 'id', how = 'left').fillna(\"none 1 0 0 1 1\")\n    sub_df = sub_df[['id', 'PredictionString']]\n    for i in range(sub_df.shape[0]):\n        if sub_df.loc[i,'PredictionString'] == \"none 1 0 0 1 1\":\n            continue\n        sub_df_split = sub_df.loc[i,'PredictionString'].split()\n        sub_df_list = []\n        for j in range(int(len(sub_df_split) / 6)):\n            sub_df_list.append('opacity')\n            sub_df_list.append(sub_df_split[6 * j + 1])\n            sub_df_list.append(sub_df_split[6 * j + 2])\n            sub_df_list.append(sub_df_split[6 * j + 3])\n            sub_df_list.append(sub_df_split[6 * j + 4])\n            sub_df_list.append(sub_df_split[6 * j + 5])\n        sub_df.loc[i,'PredictionString'] = ' '.join(sub_df_list)\n    sub_df['none'] = df_2class['none'] \n    for i in range(sub_df.shape[0]):\n        if sub_df.loc[i,'PredictionString'] != 'none 1 0 0 1 1':\n            sub_df.loc[i,'PredictionString'] = sub_df.loc[i,'PredictionString'] + ' none ' + str(sub_df.loc[i,'none']) + ' 0 0 1 1'\n    sub_df = sub_df[['id', 'PredictionString']]   \n    df_study = df_study[:study_len]\n    df_study = df_study.append(sub_df).reset_index(drop=True)\n    df_study.to_csv('/kaggle/working/submission.csv',index = False)  \n    shutil.rmtree('/kaggle/working/yolov5')","metadata":{"execution":{"iopub.status.busy":"2021-07-12T00:15:53.037932Z","iopub.execute_input":"2021-07-12T00:15:53.038264Z","iopub.status.idle":"2021-07-12T00:16:11.605027Z","shell.execute_reply.started":"2021-07-12T00:15:53.038235Z","shell.execute_reply":"2021-07-12T00:16:11.604195Z"},"trusted":true},"execution_count":null,"outputs":[]}]}